From 97fbc2fb36d756bf2abd8186d3ac41407bac01d3 Mon Sep 17 00:00:00 2001 From: Nikhil Unni Date: Tue, 6 Oct 2026 10:42:26 -0700 Subject: [PATCH] feat(coord): the durable teleport state machine; TeleportSession and RetireHost plan it (ADR 0123 B) Rebuilt onto main after the squash merge of #1590; content unchanged. Includes the review fixes (fenced teleport writes, one-transaction settlement, source reservation, move-owned sandboxes, held-source role). Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01965DMBwLXzE9baCmj1Wp8Q --- .github/scripts/detect-rebake-lanes.py | 5 +- .github/workflows/ci.yml | 20 +- cli/src/commands/host.ts | 8 +- cli/src/gen/engram/app/v1/fleet_pb.ts | 158 +- crates/engram-coordinator/src/api/admin.rs | 562 +---- .../engram-coordinator/src/api/host_http.rs | 26 +- crates/engram-coordinator/src/api/snapshot.rs | 201 +- .../src/checkpoint_retention.rs | 2 +- crates/engram-coordinator/src/dead_host.rs | 2 +- .../engram-coordinator/src/enable_scanner.rs | 4 +- crates/engram-coordinator/src/evac_resumer.rs | 928 --------- crates/engram-coordinator/src/evacuation.rs | 1421 ------------- .../engram-coordinator/src/grpc_app/fleet.rs | 81 +- .../engram-coordinator/src/host_registry.rs | 9 + crates/engram-coordinator/src/idle_evictor.rs | 39 +- crates/engram-coordinator/src/lib.rs | 13 +- .../engram-coordinator/src/live_migration.rs | 1819 ----------------- crates/engram-coordinator/src/placement.rs | 35 - .../engram-coordinator/src/queue_scanner.rs | 6 +- crates/engram-coordinator/src/session_ops.rs | 27 - .../engram-coordinator/src/session_verbs.rs | 15 +- crates/engram-coordinator/src/state.rs | 115 +- crates/engram-coordinator/src/teleport.rs | 1352 ++++++++++++ .../tests/admin_evac_live_pg.rs | 520 +---- .../tests/binding_writer_inventory.rs | 17 +- .../tests/dead_host_mock.rs | 18 + crates/engram-coordinator/tests/e2e_stack.rs | 219 +- .../engram-coordinator/tests/ha_listener.rs | 53 +- .../tests/session_ops_live_pg.rs | 29 + .../tests/support/teleport.rs | 529 +++++ .../tests/support/teleport_scenarios.rs | 712 +++++++ .../tests/teleport_live_pg.rs | 3 + crates/engram-core/src/traits/host_client.rs | 11 + crates/engram-core/src/traits/metadata.rs | 125 +- crates/engram-core/src/traits/sandbox.rs | 12 + crates/engram-core/src/types/evacuation.rs | 115 -- crates/engram-core/src/types/mod.rs | 2 - crates/engram-core/src/types/session.rs | 22 +- crates/engram-core/src/types/teleport.rs | 78 +- crates/engram-dst-cosim/src/bridge.rs | 9 + crates/engram-dst/src/invariants.rs | 52 + crates/engram-dst/src/scheduler.rs | 80 +- crates/engram-dst/src/workload.rs | 12 +- crates/engram-dst/src/world.rs | 9 + crates/engram-dst/tests/api_surface.rs | 19 +- crates/engram-dst/tests/first_sim.rs | 4 +- crates/engram-dst/tests/model_oracle.rs | 63 + crates/engram-dst/tests/recovery_oracle.rs | 82 + crates/engram-dst/tests/regression_seeds.rs | 4 +- crates/engram-dst/tests/workload_verbs.rs | 72 +- crates/engram-host-agent/src/grpc_server.rs | 24 + crates/engram-host-agent/src/host_client.rs | 8 + crates/engram-host-agent/src/lib.rs | 11 +- crates/engram-host-agent/src/migration.rs | 18 +- .../engram-host-agent/src/pooled_backend.rs | 340 ++- crates/engram-postgres/src/lib.rs | 438 +++- .../proto/engram/app/v1/fleet.proto | 43 +- .../engram-protocol/proto/host_service.proto | 2 + crates/engram-protocol/src/grpc_client.rs | 23 + crates/engram-protocol/src/wire.rs | 4 +- crates/engram-protocol/tests/wire_golden.rs | 3 +- .../engram-sandbox-firecracker/src/client.rs | 24 + crates/engram-sandbox-firecracker/src/lib.rs | 43 +- .../src/sandbox_manifest.rs | 2 + crates/engram-sandbox-process/src/lib.rs | 42 + crates/engram-sandbox-vz/src/backend.rs | 9 + crates/engram-sim/src/meta/mod.rs | 7 +- crates/engram-sim/src/meta/store_impl.rs | 479 ++++- crates/engram-sim/tests/meta_conformance.rs | 536 ++++- deploy/migrations/0122_teleport_machine.sql | 44 + ...23-durable-teleport-and-host-retirement.md | 46 + orchestrator/src/authz/policy-map.ts | 2 +- .../src/control-plane/session-events.ts | 1 + .../src/gen/engram/app/v1/fleet_pb.ts | 158 +- .../__tests__/frame-taxonomy.test.ts | 2 + orchestrator/src/routes/admin.ts | 2 +- web/src/components/SessionDiagnostics.tsx | 38 +- web/src/events.ts | 9 + .../app/v1/fleet-FleetService_connectquery.ts | 7 +- web/src/gen/engram/app/v1/fleet_pb.ts | 158 +- web/src/hooks/useTeleportSession.ts | 33 +- web/src/lib/types.ts | 2 +- web/src/sse.ts | 1 + web/src/test-utils.tsx | 4 +- 84 files changed, 5713 insertions(+), 6569 deletions(-) delete mode 100644 crates/engram-coordinator/src/evac_resumer.rs delete mode 100644 crates/engram-coordinator/src/evacuation.rs delete mode 100644 crates/engram-coordinator/src/live_migration.rs create mode 100644 crates/engram-coordinator/src/teleport.rs create mode 100644 crates/engram-coordinator/tests/support/teleport.rs create mode 100644 crates/engram-coordinator/tests/support/teleport_scenarios.rs create mode 100644 crates/engram-coordinator/tests/teleport_live_pg.rs delete mode 100644 crates/engram-core/src/types/evacuation.rs create mode 100644 deploy/migrations/0122_teleport_machine.sql diff --git a/.github/scripts/detect-rebake-lanes.py b/.github/scripts/detect-rebake-lanes.py index ad8c5f296..a90a1817c 100755 --- a/.github/scripts/detect-rebake-lanes.py +++ b/.github/scripts/detect-rebake-lanes.py @@ -521,7 +521,8 @@ def main(): "expect_two_hosts": "", "nextest_filter": ( "test(/e2e_/) " - "- test(e2e_two_host_evacuate_preserves_sentinel) " + "- test(e2e_retire_host_relocates_and_grants) " + "- test(e2e_teleport_session_honors_target_host) " "- test(e2e_claude_with_bogus_key_surfaces_anthropic_auth_error)" ), }, @@ -529,7 +530,7 @@ def main(): "variant": "teleport", "two_hosts": "1", "expect_two_hosts": "1", - "nextest_filter": "test(e2e_two_host_evacuate_preserves_sentinel)", + "nextest_filter": "test(e2e_teleport_session_honors_target_host)", }, ] e2e_orchestrator_matrix = [{ diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 2c9022111..906acf955 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -547,6 +547,7 @@ jobs: --test snapshot_blob_gc_live_pg \ --test admin_chunk_gc_live_pg \ --test admin_evac_live_pg \ + --test teleport_live_pg \ --test eviction_live_pg \ --test checkpoint_reconcile_live_pg \ --test enable_jobs_live_pg \ @@ -1796,10 +1797,7 @@ jobs: # - suite: single host. Runs every e2e_stack test EXCEPT the two-host # evacuation test and the quarantined Claude auth test. This includes # the orchestrator automation tests in the normal full posture. - # - teleport: two hosts (host-a + host-b). Runs ONLY the two-host - # evacuation test, which drives `FleetService.EvacuateSession` (the - # teleport primitive on the app-gRPC surface, ADR 0051) and asserts the - # session relocates off the source host with its disk byte-identical. + # - teleport: two hosts. Check pinned teleport, then retirement and deletion. # - orchestrator: single host. Runs only the webhook and cron automation # scenarios that call the orchestrator's public RPC and HTTP surfaces. # The detector uses this variant only for an orchestrator-only PR. @@ -1808,12 +1806,6 @@ jobs: # selects only orchestrator. Push and merge-group runs use suite and # teleport, so the full posture stays at two concurrent stacks. # - # ADR 0051 restored this two-host coverage as - # `e2e_two_host_evacuate_preserves_sentinel` (replacing the deleted REST - # `e2e_two_host_teleport_preserves_sentinel` — there is no Teleport RPC, so - # EvacuateSession is the gRPC teleport primitive; it relocates to any peer - # rather than a pinned host). - # # e2e stack IS gated now (in the CI Gate's needs). Branch protection requires # only the CI Gate, never the matrix-suffixed job names directly. strategy: @@ -2129,9 +2121,7 @@ jobs: # hosts registered (a single-host regression fails loud instead of # silently skipping). ENGRAM_EXPECT_TWO_HOSTS: "${{ matrix.expect_two_hosts }}" - # Per-variant nextest filterset: suite runs everything except the - # two-host evac test; teleport runs only it (ADR 0051 restored the - # split when EvacuateSession became the gRPC teleport primitive). + # Retirement deletes a fixture host, so it runs last in a separate invocation. E2E_NEXTEST_FILTER: "${{ matrix.nextest_filter }}" run: | cargo nextest run \ @@ -2141,6 +2131,10 @@ jobs: --no-fail-fast \ --test-threads=1 \ -E "$E2E_NEXTEST_FILTER" + if [ "${{ matrix.variant }}" = "teleport" ]; then + cargo nextest run -p engram-coordinator --test e2e_stack --run-ignored ignored-only \ + -E 'test(e2e_retire_host_relocates_and_grants)' + fi - name: Dump tilt + stack logs on failure if: failure() diff --git a/cli/src/commands/host.ts b/cli/src/commands/host.ts index 505e05eeb..6d1a595cc 100644 --- a/cli/src/commands/host.ts +++ b/cli/src/commands/host.ts @@ -2,7 +2,7 @@ * engrams host … — FleetService passthrough verbs (admin surfaces). * * `evacuate` closes the gap the Rust CLI documented ("no engram-cli surface - * for EvacuateSession") — the passthrough forwards the whole FleetService. + * for TeleportSession") — the passthrough forwards the whole FleetService. */ import type { Clients } from "../client.ts"; @@ -121,7 +121,7 @@ export async function uncordon(c: Clients, id: string, json: boolean): Promise { - const resp = await c.fleet.evacuateSession({ sessionId }).catch(failWith); - if (json) printJson({ session_id: resp.sessionId, status: resp.status }); - else console.log(`${resp.sessionId}: ${resp.status}`); + const resp = await c.fleet.teleportSession({ sessionId }).catch(failWith); + if (json) printJson({ teleport_id: resp.teleportId, kind: resp.kind, dest_host_id: resp.destHostId }); + else console.log(`${resp.teleportId}: ${resp.kind} to ${resp.destHostId}`); } diff --git a/cli/src/gen/engram/app/v1/fleet_pb.ts b/cli/src/gen/engram/app/v1/fleet_pb.ts index 7c7e01ea0..7da919176 100644 --- a/cli/src/gen/engram/app/v1/fleet_pb.ts +++ b/cli/src/gen/engram/app/v1/fleet_pb.ts @@ -12,7 +12,7 @@ import type { Message } from "@bufbuild/protobuf"; * Describes the file engram/app/v1/fleet.proto. */ export const file_engram_app_v1_fleet: GenFile = /*@__PURE__*/ - fileDesc("ChllbmdyYW0vYXBwL3YxL2ZsZWV0LnByb3RvEg1lbmdyYW0uYXBwLnYxIhIKEExpc3RIb3N0c1JlcXVlc3QiOwoRTGlzdEhvc3RzUmVzcG9uc2USJgoFaG9zdHMYASADKAsyFy5lbmdyYW0uYXBwLnYxLkhvc3RWaWV3IvEGCghIb3N0VmlldxIKCgJpZBgBIAEoCRIQCghob3N0bmFtZRgCIAEoCRIOCgZzdGF0dXMYAyABKAkSGgoSY2FwYWNpdHlfdG90YWxfbWliGAQgASgEEhkKEWNhcGFjaXR5X3VzZWRfbWliGAUgASgEEhkKEXJ1bm5pbmdfc2FuZGJveGVzGAYgASgNEhQKDHJlYWR5X2ltYWdlcxgIIAEoBBIbChNyZWFkeV9pbWFnZV9kaWdlc3RzGAkgAygJEhsKE3V0aWxfZGlza190b3RhbF9taWIYCiABKAQSGgoSdXRpbF9kaXNrX3VzZWRfbWliGAsgASgEEhoKEnV0aWxfbWVtX3RvdGFsX21pYhgMIAEoBBIZChF1dGlsX21lbV91c2VkX21pYhgNIAEoBBIUCgx1dGlsX2NwdV9wY3QYDiABKAISGQoRbGFzdF9oZWFydGJlYXRfYXQYDyABKAkSEAoIY29yZG9uZWQYECABKAgSFwoPYWxsb2NhdGFibGVfbWliGBEgASgEEhQKDHJlc2VydmVkX21pYhgSIAEoBBIQCghmcmVlX21pYhgTIAEoBBITCgt0b3RhbF92Y3B1cxgUIAEoDRIYChBjcHVfYnVkZ2V0X3ZjcHVzGBUgASgEEhYKDnJlc2VydmVkX3ZjcHVzGBYgASgEEhIKCmZyZWVfdmNwdXMYFyABKAQSGQoRdXRpbF9iYXNlX3NobV9taWIYGCABKAQSGwoTdXRpbF9wYXJrZWRfcHNzX21pYhgZIAEoBBIcChR1dGlsX3J1bm5pbmdfcHNzX21pYhgaIAEoBBIcChRmYWlsaW5nX2NhcGFiaWxpdGllcxgbIAMoCRIbChNmY19zbmFwc2hvdF92ZXJzaW9uGBwgASgJEhsKE2NhcGFiaWxpdGllc19zY2hlbWEYHSABKA0SGQoRbGl2ZV9tYXRlcmlhbGl6ZXMYHiABKA0SGQoRbGl2ZV9jYXB0dXJlX2pvYnMYHyABKA0SHwoXdXRpbF9jb21taXR0ZWRfc3dhcF9taWIYICABKAQSFAoMY29yZG9uX293bmVyGCEgASgJEjEKCnJldGlyZW1lbnQYIiABKAsyHS5lbmdyYW0uYXBwLnYxLkhvc3RSZXRpcmVtZW50SgQIBxAIUg9sb2NhbF9zbmFwc2hvdHMiIQoOR2V0SG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCSI4Cg9HZXRIb3N0UmVzcG9uc2USJQoEaG9zdBgBIAEoCzIXLmVuZ3JhbS5hcHAudjEuSG9zdFZpZXciKQoWR2V0SG9zdENvd1N0YXRlUmVxdWVzdBIPCgdob3N0X2lkGAEgASgJIlkKF0dldEhvc3RDb3dTdGF0ZVJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSLQoIc2Vzc2lvbnMYAiADKAsyGy5lbmdyYW0uYXBwLnYxLkNvd1N0YXRlVmlldyIjChBEcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkiEwoRRHJhaW5Ib3N0UmVzcG9uc2UiKAoVQWRtaW5EcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkibAoWQWRtaW5EcmFpbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEhIKCmV2YWN1YXRpbmcYAiADKAkSLQoIZmFpbHVyZXMYAyADKAsyGy5lbmdyYW0uYXBwLnYxLkRyYWluRmFpbHVyZSIxCgxEcmFpbkZhaWx1cmUSEgoKc2Vzc2lvbl9pZBgBIAEoCRINCgVlcnJvchgCIAEoCSIzChFDb3Jkb25Ib3N0UmVxdWVzdBIPCgdob3N0X2lkGAEgASgJEg0KBW93bmVyGAIgASgJIjwKF0JlZ2luSG9zdEhhbmRvZmZSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkSEAoIdHRsX3NlY3MYAiABKAQiPQoYQmVnaW5Ib3N0SGFuZG9mZlJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSEAoIYWNjZXB0ZWQYAiABKAgiNQoSQ29yZG9uSG9zdFJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSDgoGc3RhdHVzGAIgASgJIjUKE1VuY29yZG9uSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCRINCgVvd25lchgCIAEoCSI3ChRVbmNvcmRvbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEg4KBnN0YXR1cxgCIAEoCSIkChFEZWxldGVIb3N0UmVxdWVzdBIPCgdob3N0X2lkGAEgASgJIhQKEkRlbGV0ZUhvc3RSZXNwb25zZSIaChhHZXRTdG9yYWdlU3VtbWFyeVJlcXVlc3QihAIKGUdldFN0b3JhZ2VTdW1tYXJ5UmVzcG9uc2USEQoJc25hcHNob3RzGAEgASgEEhYKDnNuYXBzaG90X2J5dGVzGAIgASgEEhIKCmdjX3BlbmRpbmcYAyABKAQSGQoRdHJhY2tlZF9zYW5kYm94ZXMYBCABKAQSFAoMZGlydHlfY2h1bmtzGAUgASgEEhcKD3VuZmx1c2hlZF9ieXRlcxgGIAEoBBIYChBhdmdfbG9jYWxpdHlfcGN0GAcgASgNEioKBHJvd3MYCCADKAsyHC5lbmdyYW0uYXBwLnYxLkR1cmFiaWxpdHlSb3cSGAoQZ2NfcGVuZGluZ19leGFjdBgJIAEoCCLlAQoNRHVyYWJpbGl0eVJvdxISCgpzYW5kYm94X2lkGAEgASgJEhcKCnNlc3Npb25faWQYAiABKAlIAIgBARIPCgdob3N0X2lkGAMgASgJEhQKDGRpcnR5X2NodW5rcxgEIAEoDRITCgtkaXJ0eV9ieXRlcxgFIAEoBBITCgtiYXNlX2NodW5rcxgGIAEoDRIZChFiYXNlX2NodW5rc19sb2NhbBgHIAEoDRIaCg1sYXN0X2ZsdXNoX2F0GAggASgJSAGIAQFCDQoLX3Nlc3Npb25faWRCEAoOX2xhc3RfZmx1c2hfYXQiKQoTRmx1c2hTZXNzaW9uUmVxdWVzdBISCgpzZXNzaW9uX2lkGAEgASgJIlsKFEZsdXNoU2Vzc2lvblJlc3BvbnNlEg8KB291dGNvbWUYASABKAkSHQoQbWFuaWZlc3RfdmVyc2lvbhgCIAEoBEgAiAEBQhMKEV9tYW5pZmVzdF92ZXJzaW9uIlYKFkV2YWN1YXRlU2Vzc2lvblJlcXVlc3QSEgoKc2Vzc2lvbl9pZBgBIAEoCRIYCgt0YXJnZXRfaG9zdBgCIAEoCUgAiAEBQg4KDF90YXJnZXRfaG9zdCI9ChdFdmFjdWF0ZVNlc3Npb25SZXNwb25zZRISCgpzZXNzaW9uX2lkGAEgASgJEg4KBnN0YXR1cxgCIAEoCSIXChVHZXRGbGVldERlbWFuZFJlcXVlc3Qi8QEKFkdldEZsZWV0RGVtYW5kUmVzcG9uc2USEwoLcmVhZHlfaG9zdHMYASABKA0SGQoRc2NoZWR1bGFibGVfaG9zdHMYAiABKA0SEAoIZnJlZV9taWIYAyABKAQSEQoJdG90YWxfbWliGAQgASgEEhIKCmZyZWVfdmNwdXMYBSABKAQSEwoLdG90YWxfdmNwdXMYBiABKAQSFgoOY29yZG9uZWRfaG9zdHMYByABKA0SFwoPcXVldWVkX3Nlc3Npb25zGAggASgEEhIKCnF1ZXVlZF9taWIYCSABKAQSFAoMcXVldWVkX3ZjcHVzGAogASgEIkkKDkNodW5rR2NSZXF1ZXN0Eg8KB2RyeV9ydW4YASABKAgSFwoKZ3JhY2Vfc2VjcxgCIAEoBEgAiAEBQg0KC19ncmFjZV9zZWNzIowCCg9DaHVua0djUmVzcG9uc2USFQoNbGlzdGVkX2NodW5rcxgBIAEoBBIWCg5tYWxmb3JtZWRfa2V5cxgCIAEoBBIUCgxwaW5fc2V0X3NpemUYAyABKAQSGQoRY2FuZGlkYXRlc19tYXJrZWQYBCABKAQSGAoQcHJvbW90ZWRfZGVsZXRlcxgHIAEoBBIdChVwcm9tb3RlX2RlbGV0ZV9lcnJvcnMYCCABKAQSEgoKZ3JhY2Vfc2VjcxgJIAEoBBIYChBnZW5lcmF0aW9uX21vdmVkGAogASgIEhcKCm1hcmtfZXJyb3IYCyABKAlIAIgBAUINCgtfbWFya19lcnJvckoECAUQBkoECAYQByJKCg9CdW5kbGVHY1JlcXVlc3QSDwoHZHJ5X3J1bhgBIAEoCBIXCgpncmFjZV9zZWNzGAIgASgESACIAQFCDQoLX2dyYWNlX3NlY3MiwwEKEEJ1bmRsZUdjUmVzcG9uc2USDgoGbGlzdGVkGAEgASgEEhQKDHBpbl9zZXRfc2l6ZRgCIAEoBBIZChFjYW5kaWRhdGVzX21hcmtlZBgDIAEoBBIYChBwcm9tb3RlZF9kZWxldGVzGAQgASgEEh0KFXByb21vdGVfZGVsZXRlX2Vycm9ycxgFIAEoBBIVCg1yZXN0YXJ0X2NvdW50GAYgASgNEh4KFnByb21vdGVfcmVwaW5uZWRfc2tpcHMYByABKAQiUAoVU25hcHNob3RCbG9iR2NSZXF1ZXN0Eg8KB2RyeV9ydW4YASABKAgSFwoKZ3JhY2Vfc2VjcxgCIAEoBEgAiAEBQg0KC19ncmFjZV9zZWNzItwBChZTbmFwc2hvdEJsb2JHY1Jlc3BvbnNlEg4KBmxpc3RlZBgBIAEoBBIRCgltYWxmb3JtZWQYAiABKAQSFAoMcGluX3NldF9zaXplGAMgASgEEhkKEWNhbmRpZGF0ZXNfbWFya2VkGAQgASgEEhgKEHByb21vdGVkX2RlbGV0ZXMYBSABKAQSHgoWcHJvbW90ZV9yZXBpbm5lZF9za2lwcxgGIAEoBBIdChVwcm9tb3RlX2RlbGV0ZV9lcnJvcnMYByABKAQSFQoNcmVzdGFydF9jb3VudBgIIAEoDSJDChFSZXRpcmVIb3N0UmVxdWVzdBIPCgdob3N0X2lkGAEgASgJEg0KBW93bmVyGAIgASgJEg4KBnJlYXNvbhgDIAEoCSJzChJSZXRpcmVIb3N0UmVzcG9uc2USDwoHaG9zdF9pZBgBIAEoCRIxCgpyZXRpcmVtZW50GAIgASgLMh0uZW5ncmFtLmFwcC52MS5Ib3N0UmV0aXJlbWVudBIZChF0ZWxlcG9ydHNfcGxhbm5lZBgDIAEoDSJuCg5Ib3N0UmV0aXJlbWVudBIUCgxyZXF1ZXN0ZWRfYXQYASABKAkSEgoKcmV0aXJlZF9hdBgCIAEoCRIyCghibG9ja2VycxgDIAMoCzIgLmVuZ3JhbS5hcHAudjEuUmV0aXJlbWVudEJsb2NrZXIiMAoRUmV0aXJlbWVudEJsb2NrZXISDAoEa2luZBgBIAEoCRINCgVjb3VudBgCIAEoBDLoCwoMRmxlZXRTZXJ2aWNlEk4KCUxpc3RIb3N0cxIfLmVuZ3JhbS5hcHAudjEuTGlzdEhvc3RzUmVxdWVzdBogLmVuZ3JhbS5hcHAudjEuTGlzdEhvc3RzUmVzcG9uc2USSAoHR2V0SG9zdBIdLmVuZ3JhbS5hcHAudjEuR2V0SG9zdFJlcXVlc3QaHi5lbmdyYW0uYXBwLnYxLkdldEhvc3RSZXNwb25zZRJRCgpSZXRpcmVIb3N0EiAuZW5ncmFtLmFwcC52MS5SZXRpcmVIb3N0UmVxdWVzdBohLmVuZ3JhbS5hcHAudjEuUmV0aXJlSG9zdFJlc3BvbnNlEmAKD0dldEhvc3RDb3dTdGF0ZRIlLmVuZ3JhbS5hcHAudjEuR2V0SG9zdENvd1N0YXRlUmVxdWVzdBomLmVuZ3JhbS5hcHAudjEuR2V0SG9zdENvd1N0YXRlUmVzcG9uc2USTgoJRHJhaW5Ib3N0Eh8uZW5ncmFtLmFwcC52MS5EcmFpbkhvc3RSZXF1ZXN0GiAuZW5ncmFtLmFwcC52MS5EcmFpbkhvc3RSZXNwb25zZRJdCg5BZG1pbkRyYWluSG9zdBIkLmVuZ3JhbS5hcHAudjEuQWRtaW5EcmFpbkhvc3RSZXF1ZXN0GiUuZW5ncmFtLmFwcC52MS5BZG1pbkRyYWluSG9zdFJlc3BvbnNlElEKCkNvcmRvbkhvc3QSIC5lbmdyYW0uYXBwLnYxLkNvcmRvbkhvc3RSZXF1ZXN0GiEuZW5ncmFtLmFwcC52MS5Db3Jkb25Ib3N0UmVzcG9uc2USYwoQQmVnaW5Ib3N0SGFuZG9mZhImLmVuZ3JhbS5hcHAudjEuQmVnaW5Ib3N0SGFuZG9mZlJlcXVlc3QaJy5lbmdyYW0uYXBwLnYxLkJlZ2luSG9zdEhhbmRvZmZSZXNwb25zZRJXCgxVbmNvcmRvbkhvc3QSIi5lbmdyYW0uYXBwLnYxLlVuY29yZG9uSG9zdFJlcXVlc3QaIy5lbmdyYW0uYXBwLnYxLlVuY29yZG9uSG9zdFJlc3BvbnNlElEKCkRlbGV0ZUhvc3QSIC5lbmdyYW0uYXBwLnYxLkRlbGV0ZUhvc3RSZXF1ZXN0GiEuZW5ncmFtLmFwcC52MS5EZWxldGVIb3N0UmVzcG9uc2USZgoRR2V0U3RvcmFnZVN1bW1hcnkSJy5lbmdyYW0uYXBwLnYxLkdldFN0b3JhZ2VTdW1tYXJ5UmVxdWVzdBooLmVuZ3JhbS5hcHAudjEuR2V0U3RvcmFnZVN1bW1hcnlSZXNwb25zZRJXCgxGbHVzaFNlc3Npb24SIi5lbmdyYW0uYXBwLnYxLkZsdXNoU2Vzc2lvblJlcXVlc3QaIy5lbmdyYW0uYXBwLnYxLkZsdXNoU2Vzc2lvblJlc3BvbnNlEmAKD0V2YWN1YXRlU2Vzc2lvbhIlLmVuZ3JhbS5hcHAudjEuRXZhY3VhdGVTZXNzaW9uUmVxdWVzdBomLmVuZ3JhbS5hcHAudjEuRXZhY3VhdGVTZXNzaW9uUmVzcG9uc2USSAoHQ2h1bmtHYxIdLmVuZ3JhbS5hcHAudjEuQ2h1bmtHY1JlcXVlc3QaHi5lbmdyYW0uYXBwLnYxLkNodW5rR2NSZXNwb25zZRJLCghCdW5kbGVHYxIeLmVuZ3JhbS5hcHAudjEuQnVuZGxlR2NSZXF1ZXN0Gh8uZW5ncmFtLmFwcC52MS5CdW5kbGVHY1Jlc3BvbnNlEl0KDlNuYXBzaG90QmxvYkdjEiQuZW5ncmFtLmFwcC52MS5TbmFwc2hvdEJsb2JHY1JlcXVlc3QaJS5lbmdyYW0uYXBwLnYxLlNuYXBzaG90QmxvYkdjUmVzcG9uc2USXQoOR2V0RmxlZXREZW1hbmQSJC5lbmdyYW0uYXBwLnYxLkdldEZsZWV0RGVtYW5kUmVxdWVzdBolLmVuZ3JhbS5hcHAudjEuR2V0RmxlZXREZW1hbmRSZXNwb25zZWIGcHJvdG8z", [file_engram_app_v1_session]); + fileDesc("ChllbmdyYW0vYXBwL3YxL2ZsZWV0LnByb3RvEg1lbmdyYW0uYXBwLnYxIhIKEExpc3RIb3N0c1JlcXVlc3QiOwoRTGlzdEhvc3RzUmVzcG9uc2USJgoFaG9zdHMYASADKAsyFy5lbmdyYW0uYXBwLnYxLkhvc3RWaWV3IvEGCghIb3N0VmlldxIKCgJpZBgBIAEoCRIQCghob3N0bmFtZRgCIAEoCRIOCgZzdGF0dXMYAyABKAkSGgoSY2FwYWNpdHlfdG90YWxfbWliGAQgASgEEhkKEWNhcGFjaXR5X3VzZWRfbWliGAUgASgEEhkKEXJ1bm5pbmdfc2FuZGJveGVzGAYgASgNEhQKDHJlYWR5X2ltYWdlcxgIIAEoBBIbChNyZWFkeV9pbWFnZV9kaWdlc3RzGAkgAygJEhsKE3V0aWxfZGlza190b3RhbF9taWIYCiABKAQSGgoSdXRpbF9kaXNrX3VzZWRfbWliGAsgASgEEhoKEnV0aWxfbWVtX3RvdGFsX21pYhgMIAEoBBIZChF1dGlsX21lbV91c2VkX21pYhgNIAEoBBIUCgx1dGlsX2NwdV9wY3QYDiABKAISGQoRbGFzdF9oZWFydGJlYXRfYXQYDyABKAkSEAoIY29yZG9uZWQYECABKAgSFwoPYWxsb2NhdGFibGVfbWliGBEgASgEEhQKDHJlc2VydmVkX21pYhgSIAEoBBIQCghmcmVlX21pYhgTIAEoBBITCgt0b3RhbF92Y3B1cxgUIAEoDRIYChBjcHVfYnVkZ2V0X3ZjcHVzGBUgASgEEhYKDnJlc2VydmVkX3ZjcHVzGBYgASgEEhIKCmZyZWVfdmNwdXMYFyABKAQSGQoRdXRpbF9iYXNlX3NobV9taWIYGCABKAQSGwoTdXRpbF9wYXJrZWRfcHNzX21pYhgZIAEoBBIcChR1dGlsX3J1bm5pbmdfcHNzX21pYhgaIAEoBBIcChRmYWlsaW5nX2NhcGFiaWxpdGllcxgbIAMoCRIbChNmY19zbmFwc2hvdF92ZXJzaW9uGBwgASgJEhsKE2NhcGFiaWxpdGllc19zY2hlbWEYHSABKA0SGQoRbGl2ZV9tYXRlcmlhbGl6ZXMYHiABKA0SGQoRbGl2ZV9jYXB0dXJlX2pvYnMYHyABKA0SHwoXdXRpbF9jb21taXR0ZWRfc3dhcF9taWIYICABKAQSFAoMY29yZG9uX293bmVyGCEgASgJEjEKCnJldGlyZW1lbnQYIiABKAsyHS5lbmdyYW0uYXBwLnYxLkhvc3RSZXRpcmVtZW50SgQIBxAIUg9sb2NhbF9zbmFwc2hvdHMiIQoOR2V0SG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCSI4Cg9HZXRIb3N0UmVzcG9uc2USJQoEaG9zdBgBIAEoCzIXLmVuZ3JhbS5hcHAudjEuSG9zdFZpZXciKQoWR2V0SG9zdENvd1N0YXRlUmVxdWVzdBIPCgdob3N0X2lkGAEgASgJIlkKF0dldEhvc3RDb3dTdGF0ZVJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSLQoIc2Vzc2lvbnMYAiADKAsyGy5lbmdyYW0uYXBwLnYxLkNvd1N0YXRlVmlldyIjChBEcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkiEwoRRHJhaW5Ib3N0UmVzcG9uc2UiKAoVQWRtaW5EcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkiXgoWQWRtaW5EcmFpbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEg8KB3BsYW5uZWQYAiADKAkSEQoJZGVzY2VuZGVkGAMgAygJEg8KB3NraXBwZWQYBCABKA0iMwoRQ29yZG9uSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCRINCgVvd25lchgCIAEoCSI8ChdCZWdpbkhvc3RIYW5kb2ZmUmVxdWVzdBIPCgdob3N0X2lkGAEgASgJEhAKCHR0bF9zZWNzGAIgASgEIj0KGEJlZ2luSG9zdEhhbmRvZmZSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEhAKCGFjY2VwdGVkGAIgASgIIjUKEkNvcmRvbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEg4KBnN0YXR1cxgCIAEoCSI1ChNVbmNvcmRvbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkSDQoFb3duZXIYAiABKAkiNwoUVW5jb3Jkb25Ib3N0UmVzcG9uc2USDwoHaG9zdF9pZBgBIAEoCRIOCgZzdGF0dXMYAiABKAkiJAoRRGVsZXRlSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCSIUChJEZWxldGVIb3N0UmVzcG9uc2UiGgoYR2V0U3RvcmFnZVN1bW1hcnlSZXF1ZXN0IoQCChlHZXRTdG9yYWdlU3VtbWFyeVJlc3BvbnNlEhEKCXNuYXBzaG90cxgBIAEoBBIWCg5zbmFwc2hvdF9ieXRlcxgCIAEoBBISCgpnY19wZW5kaW5nGAMgASgEEhkKEXRyYWNrZWRfc2FuZGJveGVzGAQgASgEEhQKDGRpcnR5X2NodW5rcxgFIAEoBBIXCg91bmZsdXNoZWRfYnl0ZXMYBiABKAQSGAoQYXZnX2xvY2FsaXR5X3BjdBgHIAEoDRIqCgRyb3dzGAggAygLMhwuZW5ncmFtLmFwcC52MS5EdXJhYmlsaXR5Um93EhgKEGdjX3BlbmRpbmdfZXhhY3QYCSABKAgi5QEKDUR1cmFiaWxpdHlSb3cSEgoKc2FuZGJveF9pZBgBIAEoCRIXCgpzZXNzaW9uX2lkGAIgASgJSACIAQESDwoHaG9zdF9pZBgDIAEoCRIUCgxkaXJ0eV9jaHVua3MYBCABKA0SEwoLZGlydHlfYnl0ZXMYBSABKAQSEwoLYmFzZV9jaHVua3MYBiABKA0SGQoRYmFzZV9jaHVua3NfbG9jYWwYByABKA0SGgoNbGFzdF9mbHVzaF9hdBgIIAEoCUgBiAEBQg0KC19zZXNzaW9uX2lkQhAKDl9sYXN0X2ZsdXNoX2F0IikKE0ZsdXNoU2Vzc2lvblJlcXVlc3QSEgoKc2Vzc2lvbl9pZBgBIAEoCSJbChRGbHVzaFNlc3Npb25SZXNwb25zZRIPCgdvdXRjb21lGAEgASgJEh0KEG1hbmlmZXN0X3ZlcnNpb24YAiABKARIAIgBAUITChFfbWFuaWZlc3RfdmVyc2lvbiJWChZUZWxlcG9ydFNlc3Npb25SZXF1ZXN0EhIKCnNlc3Npb25faWQYASABKAkSGAoLdGFyZ2V0X2hvc3QYAiABKAlIAIgBAUIOCgxfdGFyZ2V0X2hvc3QiUgoXVGVsZXBvcnRTZXNzaW9uUmVzcG9uc2USEwoLdGVsZXBvcnRfaWQYASABKAkSDAoEa2luZBgCIAEoCRIUCgxkZXN0X2hvc3RfaWQYAyABKAkiFwoVR2V0RmxlZXREZW1hbmRSZXF1ZXN0IvEBChZHZXRGbGVldERlbWFuZFJlc3BvbnNlEhMKC3JlYWR5X2hvc3RzGAEgASgNEhkKEXNjaGVkdWxhYmxlX2hvc3RzGAIgASgNEhAKCGZyZWVfbWliGAMgASgEEhEKCXRvdGFsX21pYhgEIAEoBBISCgpmcmVlX3ZjcHVzGAUgASgEEhMKC3RvdGFsX3ZjcHVzGAYgASgEEhYKDmNvcmRvbmVkX2hvc3RzGAcgASgNEhcKD3F1ZXVlZF9zZXNzaW9ucxgIIAEoBBISCgpxdWV1ZWRfbWliGAkgASgEEhQKDHF1ZXVlZF92Y3B1cxgKIAEoBCJJCg5DaHVua0djUmVxdWVzdBIPCgdkcnlfcnVuGAEgASgIEhcKCmdyYWNlX3NlY3MYAiABKARIAIgBAUINCgtfZ3JhY2Vfc2VjcyKMAgoPQ2h1bmtHY1Jlc3BvbnNlEhUKDWxpc3RlZF9jaHVua3MYASABKAQSFgoObWFsZm9ybWVkX2tleXMYAiABKAQSFAoMcGluX3NldF9zaXplGAMgASgEEhkKEWNhbmRpZGF0ZXNfbWFya2VkGAQgASgEEhgKEHByb21vdGVkX2RlbGV0ZXMYByABKAQSHQoVcHJvbW90ZV9kZWxldGVfZXJyb3JzGAggASgEEhIKCmdyYWNlX3NlY3MYCSABKAQSGAoQZ2VuZXJhdGlvbl9tb3ZlZBgKIAEoCBIXCgptYXJrX2Vycm9yGAsgASgJSACIAQFCDQoLX21hcmtfZXJyb3JKBAgFEAZKBAgGEAciSgoPQnVuZGxlR2NSZXF1ZXN0Eg8KB2RyeV9ydW4YASABKAgSFwoKZ3JhY2Vfc2VjcxgCIAEoBEgAiAEBQg0KC19ncmFjZV9zZWNzIsMBChBCdW5kbGVHY1Jlc3BvbnNlEg4KBmxpc3RlZBgBIAEoBBIUCgxwaW5fc2V0X3NpemUYAiABKAQSGQoRY2FuZGlkYXRlc19tYXJrZWQYAyABKAQSGAoQcHJvbW90ZWRfZGVsZXRlcxgEIAEoBBIdChVwcm9tb3RlX2RlbGV0ZV9lcnJvcnMYBSABKAQSFQoNcmVzdGFydF9jb3VudBgGIAEoDRIeChZwcm9tb3RlX3JlcGlubmVkX3NraXBzGAcgASgEIlAKFVNuYXBzaG90QmxvYkdjUmVxdWVzdBIPCgdkcnlfcnVuGAEgASgIEhcKCmdyYWNlX3NlY3MYAiABKARIAIgBAUINCgtfZ3JhY2Vfc2VjcyLcAQoWU25hcHNob3RCbG9iR2NSZXNwb25zZRIOCgZsaXN0ZWQYASABKAQSEQoJbWFsZm9ybWVkGAIgASgEEhQKDHBpbl9zZXRfc2l6ZRgDIAEoBBIZChFjYW5kaWRhdGVzX21hcmtlZBgEIAEoBBIYChBwcm9tb3RlZF9kZWxldGVzGAUgASgEEh4KFnByb21vdGVfcmVwaW5uZWRfc2tpcHMYBiABKAQSHQoVcHJvbW90ZV9kZWxldGVfZXJyb3JzGAcgASgEEhUKDXJlc3RhcnRfY291bnQYCCABKA0iQwoRUmV0aXJlSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCRINCgVvd25lchgCIAEoCRIOCgZyZWFzb24YAyABKAkicwoSUmV0aXJlSG9zdFJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSMQoKcmV0aXJlbWVudBgCIAEoCzIdLmVuZ3JhbS5hcHAudjEuSG9zdFJldGlyZW1lbnQSGQoRdGVsZXBvcnRzX3BsYW5uZWQYAyABKA0ibgoOSG9zdFJldGlyZW1lbnQSFAoMcmVxdWVzdGVkX2F0GAEgASgJEhIKCnJldGlyZWRfYXQYAiABKAkSMgoIYmxvY2tlcnMYAyADKAsyIC5lbmdyYW0uYXBwLnYxLlJldGlyZW1lbnRCbG9ja2VyIjAKEVJldGlyZW1lbnRCbG9ja2VyEgwKBGtpbmQYASABKAkSDQoFY291bnQYAiABKAQy6AsKDEZsZWV0U2VydmljZRJOCglMaXN0SG9zdHMSHy5lbmdyYW0uYXBwLnYxLkxpc3RIb3N0c1JlcXVlc3QaIC5lbmdyYW0uYXBwLnYxLkxpc3RIb3N0c1Jlc3BvbnNlEkgKB0dldEhvc3QSHS5lbmdyYW0uYXBwLnYxLkdldEhvc3RSZXF1ZXN0Gh4uZW5ncmFtLmFwcC52MS5HZXRIb3N0UmVzcG9uc2USUQoKUmV0aXJlSG9zdBIgLmVuZ3JhbS5hcHAudjEuUmV0aXJlSG9zdFJlcXVlc3QaIS5lbmdyYW0uYXBwLnYxLlJldGlyZUhvc3RSZXNwb25zZRJgCg9HZXRIb3N0Q293U3RhdGUSJS5lbmdyYW0uYXBwLnYxLkdldEhvc3RDb3dTdGF0ZVJlcXVlc3QaJi5lbmdyYW0uYXBwLnYxLkdldEhvc3RDb3dTdGF0ZVJlc3BvbnNlEk4KCURyYWluSG9zdBIfLmVuZ3JhbS5hcHAudjEuRHJhaW5Ib3N0UmVxdWVzdBogLmVuZ3JhbS5hcHAudjEuRHJhaW5Ib3N0UmVzcG9uc2USXQoOQWRtaW5EcmFpbkhvc3QSJC5lbmdyYW0uYXBwLnYxLkFkbWluRHJhaW5Ib3N0UmVxdWVzdBolLmVuZ3JhbS5hcHAudjEuQWRtaW5EcmFpbkhvc3RSZXNwb25zZRJRCgpDb3Jkb25Ib3N0EiAuZW5ncmFtLmFwcC52MS5Db3Jkb25Ib3N0UmVxdWVzdBohLmVuZ3JhbS5hcHAudjEuQ29yZG9uSG9zdFJlc3BvbnNlEmMKEEJlZ2luSG9zdEhhbmRvZmYSJi5lbmdyYW0uYXBwLnYxLkJlZ2luSG9zdEhhbmRvZmZSZXF1ZXN0GicuZW5ncmFtLmFwcC52MS5CZWdpbkhvc3RIYW5kb2ZmUmVzcG9uc2USVwoMVW5jb3Jkb25Ib3N0EiIuZW5ncmFtLmFwcC52MS5VbmNvcmRvbkhvc3RSZXF1ZXN0GiMuZW5ncmFtLmFwcC52MS5VbmNvcmRvbkhvc3RSZXNwb25zZRJRCgpEZWxldGVIb3N0EiAuZW5ncmFtLmFwcC52MS5EZWxldGVIb3N0UmVxdWVzdBohLmVuZ3JhbS5hcHAudjEuRGVsZXRlSG9zdFJlc3BvbnNlEmYKEUdldFN0b3JhZ2VTdW1tYXJ5EicuZW5ncmFtLmFwcC52MS5HZXRTdG9yYWdlU3VtbWFyeVJlcXVlc3QaKC5lbmdyYW0uYXBwLnYxLkdldFN0b3JhZ2VTdW1tYXJ5UmVzcG9uc2USVwoMRmx1c2hTZXNzaW9uEiIuZW5ncmFtLmFwcC52MS5GbHVzaFNlc3Npb25SZXF1ZXN0GiMuZW5ncmFtLmFwcC52MS5GbHVzaFNlc3Npb25SZXNwb25zZRJgCg9UZWxlcG9ydFNlc3Npb24SJS5lbmdyYW0uYXBwLnYxLlRlbGVwb3J0U2Vzc2lvblJlcXVlc3QaJi5lbmdyYW0uYXBwLnYxLlRlbGVwb3J0U2Vzc2lvblJlc3BvbnNlEkgKB0NodW5rR2MSHS5lbmdyYW0uYXBwLnYxLkNodW5rR2NSZXF1ZXN0Gh4uZW5ncmFtLmFwcC52MS5DaHVua0djUmVzcG9uc2USSwoIQnVuZGxlR2MSHi5lbmdyYW0uYXBwLnYxLkJ1bmRsZUdjUmVxdWVzdBofLmVuZ3JhbS5hcHAudjEuQnVuZGxlR2NSZXNwb25zZRJdCg5TbmFwc2hvdEJsb2JHYxIkLmVuZ3JhbS5hcHAudjEuU25hcHNob3RCbG9iR2NSZXF1ZXN0GiUuZW5ncmFtLmFwcC52MS5TbmFwc2hvdEJsb2JHY1Jlc3BvbnNlEl0KDkdldEZsZWV0RGVtYW5kEiQuZW5ncmFtLmFwcC52MS5HZXRGbGVldERlbWFuZFJlcXVlc3QaJS5lbmdyYW0uYXBwLnYxLkdldEZsZWV0RGVtYW5kUmVzcG9uc2ViBnByb3RvMw", [file_engram_app_v1_session]); /** * @generated from message engram.app.v1.ListHostsRequest @@ -421,10 +421,7 @@ export const AdminDrainHostRequestSchema: GenMessage = /* messageDesc(file_engram_app_v1_fleet, 9); /** - * Mirrors api/admin.rs DrainHostResponse (renamed: this RPC carries the - * admin cordon+evacuate semantics; the soft status flip above keeps the - * plain DrainHost name). HTTP returns 202 — the evac_resumer scanner is - * the actual deliverable. + * Admission plans return immediately; the teleport machine drives each move. * * @generated from message engram.app.v1.AdminDrainHostResponse */ @@ -435,17 +432,19 @@ export type AdminDrainHostResponse = Message<"engram.app.v1.AdminDrainHostRespon hostId: string; /** - * Session ids now marked Evacuating; the scanner resumes each on a - * peer host within its sweep interval. - * - * @generated from field: repeated string evacuating = 2; + * @generated from field: repeated string planned = 2; + */ + planned: string[]; + + /** + * @generated from field: repeated string descended = 3; */ - evacuating: string[]; + descended: string[]; /** - * @generated from field: repeated engram.app.v1.DrainFailure failures = 3; + * @generated from field: uint32 skipped = 4; */ - failures: DrainFailure[]; + skipped: number; }; /** @@ -455,30 +454,6 @@ export type AdminDrainHostResponse = Message<"engram.app.v1.AdminDrainHostRespon export const AdminDrainHostResponseSchema: GenMessage = /*@__PURE__*/ messageDesc(file_engram_app_v1_fleet, 10); -/** - * @generated from message engram.app.v1.DrainFailure - */ -export type DrainFailure = Message<"engram.app.v1.DrainFailure"> & { - /** - * @generated from field: string session_id = 1; - */ - sessionId: string; - - /** - * Human-readable diagnostic; not machine-parsed, not stable. - * - * @generated from field: string error = 2; - */ - error: string; -}; - -/** - * Describes the message engram.app.v1.DrainFailure. - * Use `create(DrainFailureSchema)` to create a new message. - */ -export const DrainFailureSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 11); - /** * @generated from message engram.app.v1.CordonHostRequest */ @@ -502,7 +477,7 @@ export type CordonHostRequest = Message<"engram.app.v1.CordonHostRequest"> & { * Use `create(CordonHostRequestSchema)` to create a new message. */ export const CordonHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 12); + messageDesc(file_engram_app_v1_fleet, 11); /** * ADR 0116 A-D2: declare a planned handoff — extend the host's @@ -532,7 +507,7 @@ export type BeginHostHandoffRequest = Message<"engram.app.v1.BeginHostHandoffReq * Use `create(BeginHostHandoffRequestSchema)` to create a new message. */ export const BeginHostHandoffRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 13); + messageDesc(file_engram_app_v1_fleet, 12); /** * @generated from message engram.app.v1.BeginHostHandoffResponse @@ -558,7 +533,7 @@ export type BeginHostHandoffResponse = Message<"engram.app.v1.BeginHostHandoffRe * Use `create(BeginHostHandoffResponseSchema)` to create a new message. */ export const BeginHostHandoffResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 14); + messageDesc(file_engram_app_v1_fleet, 13); /** * @generated from message engram.app.v1.CordonHostResponse @@ -582,7 +557,7 @@ export type CordonHostResponse = Message<"engram.app.v1.CordonHostResponse"> & { * Use `create(CordonHostResponseSchema)` to create a new message. */ export const CordonHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 15); + messageDesc(file_engram_app_v1_fleet, 14); /** * @generated from message engram.app.v1.UncordonHostRequest @@ -606,7 +581,7 @@ export type UncordonHostRequest = Message<"engram.app.v1.UncordonHostRequest"> & * Use `create(UncordonHostRequestSchema)` to create a new message. */ export const UncordonHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 16); + messageDesc(file_engram_app_v1_fleet, 15); /** * Same Rust type as CordonHostResponse (CordonResponse); a distinct @@ -633,7 +608,7 @@ export type UncordonHostResponse = Message<"engram.app.v1.UncordonHostResponse"> * Use `create(UncordonHostResponseSchema)` to create a new message. */ export const UncordonHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 17); + messageDesc(file_engram_app_v1_fleet, 16); /** * @generated from message engram.app.v1.DeleteHostRequest @@ -650,7 +625,7 @@ export type DeleteHostRequest = Message<"engram.app.v1.DeleteHostRequest"> & { * Use `create(DeleteHostRequestSchema)` to create a new message. */ export const DeleteHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 18); + messageDesc(file_engram_app_v1_fleet, 17); /** * Mirrors the old `DELETE /api/admin/hosts/:id` (204 No Content) — the @@ -667,7 +642,7 @@ export type DeleteHostResponse = Message<"engram.app.v1.DeleteHostResponse"> & { * Use `create(DeleteHostResponseSchema)` to create a new message. */ export const DeleteHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 19); + messageDesc(file_engram_app_v1_fleet, 18); /** * @generated from message engram.app.v1.GetStorageSummaryRequest @@ -680,7 +655,7 @@ export type GetStorageSummaryRequest = Message<"engram.app.v1.GetStorageSummaryR * Use `create(GetStorageSummaryRequestSchema)` to create a new message. */ export const GetStorageSummaryRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 20); + messageDesc(file_engram_app_v1_fleet, 19); /** * Mirrors api/storage.rs StorageSummaryResponse (ADR 0029): fleet-wide @@ -760,7 +735,7 @@ export type GetStorageSummaryResponse = Message<"engram.app.v1.GetStorageSummary * Use `create(GetStorageSummaryResponseSchema)` to create a new message. */ export const GetStorageSummaryResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 21); + messageDesc(file_engram_app_v1_fleet, 20); /** * One per-sandbox row of the durability ledger. @@ -824,7 +799,7 @@ export type DurabilityRow = Message<"engram.app.v1.DurabilityRow"> & { * Use `create(DurabilityRowSchema)` to create a new message. */ export const DurabilityRowSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 22); + messageDesc(file_engram_app_v1_fleet, 21); /** * @generated from message engram.app.v1.FlushSessionRequest @@ -841,7 +816,7 @@ export type FlushSessionRequest = Message<"engram.app.v1.FlushSessionRequest"> & * Use `create(FlushSessionRequestSchema)` to create a new message. */ export const FlushSessionRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 23); + messageDesc(file_engram_app_v1_fleet, 22); /** * Mirrors api/admin.rs FlushNowResult. @@ -873,64 +848,56 @@ export type FlushSessionResponse = Message<"engram.app.v1.FlushSessionResponse"> * Use `create(FlushSessionResponseSchema)` to create a new message. */ export const FlushSessionResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 24); + messageDesc(file_engram_app_v1_fleet, 23); /** - * @generated from message engram.app.v1.EvacuateSessionRequest + * @generated from message engram.app.v1.TeleportSessionRequest */ -export type EvacuateSessionRequest = Message<"engram.app.v1.EvacuateSessionRequest"> & { +export type TeleportSessionRequest = Message<"engram.app.v1.TeleportSessionRequest"> & { /** * @generated from field: string session_id = 1; */ sessionId: string; /** - * Reserved for a future operator override; carried on today's HTTP - * body (api/admin.rs EvacuateSessionRequest.target_host) but IGNORED - * by the handler — the scanner picks any non-source host via the - * standard policy. - * * @generated from field: optional string target_host = 2; */ targetHost?: string; }; /** - * Describes the message engram.app.v1.EvacuateSessionRequest. - * Use `create(EvacuateSessionRequestSchema)` to create a new message. + * Describes the message engram.app.v1.TeleportSessionRequest. + * Use `create(TeleportSessionRequestSchema)` to create a new message. */ -export const EvacuateSessionRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 25); +export const TeleportSessionRequestSchema: GenMessage = /*@__PURE__*/ + messageDesc(file_engram_app_v1_fleet, 24); /** - * Mirrors api/admin.rs EvacuateSessionResponse. HTTP returns 202; the - * evac_resumer scanner completes the transition. - * - * @generated from message engram.app.v1.EvacuateSessionResponse + * @generated from message engram.app.v1.TeleportSessionResponse */ -export type EvacuateSessionResponse = Message<"engram.app.v1.EvacuateSessionResponse"> & { +export type TeleportSessionResponse = Message<"engram.app.v1.TeleportSessionResponse"> & { /** - * @generated from field: string session_id = 1; + * @generated from field: string teleport_id = 1; */ - sessionId: string; + teleportId: string; /** - * Always "evacuating" on success — the session is paused, - * snapshotted, and the scanner will resume it on a peer within the - * next sweep interval (≤10s default). Watch the session's event - * stream for the Evacuating → Created → Active chain. - * - * @generated from field: string status = 2; + * @generated from field: string kind = 2; */ - status: string; + kind: string; + + /** + * @generated from field: string dest_host_id = 3; + */ + destHostId: string; }; /** - * Describes the message engram.app.v1.EvacuateSessionResponse. - * Use `create(EvacuateSessionResponseSchema)` to create a new message. + * Describes the message engram.app.v1.TeleportSessionResponse. + * Use `create(TeleportSessionResponseSchema)` to create a new message. */ -export const EvacuateSessionResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 26); +export const TeleportSessionResponseSchema: GenMessage = /*@__PURE__*/ + messageDesc(file_engram_app_v1_fleet, 25); /** * @generated from message engram.app.v1.GetFleetDemandRequest @@ -943,7 +910,7 @@ export type GetFleetDemandRequest = Message<"engram.app.v1.GetFleetDemandRequest * Use `create(GetFleetDemandRequestSchema)` to create a new message. */ export const GetFleetDemandRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 27); + messageDesc(file_engram_app_v1_fleet, 26); /** * Mirrors api/admin.rs FleetDemandResponse (ADR 0044 K4 / 0047 / 0048). The @@ -1028,7 +995,7 @@ export type GetFleetDemandResponse = Message<"engram.app.v1.GetFleetDemandRespon * Use `create(GetFleetDemandResponseSchema)` to create a new message. */ export const GetFleetDemandResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 28); + messageDesc(file_engram_app_v1_fleet, 27); /** * @generated from message engram.app.v1.ChunkGcRequest @@ -1065,7 +1032,7 @@ export type ChunkGcRequest = Message<"engram.app.v1.ChunkGcRequest"> & { * Use `create(ChunkGcRequestSchema)` to create a new message. */ export const ChunkGcRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 29); + messageDesc(file_engram_app_v1_fleet, 28); /** * Mirrors api/admin.rs ChunkGcSweepResult. @@ -1144,7 +1111,7 @@ export type ChunkGcResponse = Message<"engram.app.v1.ChunkGcResponse"> & { * Use `create(ChunkGcResponseSchema)` to create a new message. */ export const ChunkGcResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 30); + messageDesc(file_engram_app_v1_fleet, 29); /** * @generated from message engram.app.v1.BundleGcRequest @@ -1169,7 +1136,7 @@ export type BundleGcRequest = Message<"engram.app.v1.BundleGcRequest"> & { * Use `create(BundleGcRequestSchema)` to create a new message. */ export const BundleGcRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 31); + messageDesc(file_engram_app_v1_fleet, 30); /** * Mirrors bundle_gc.rs BundleSweepReport (ADR 0035 §5). Unlike the @@ -1226,7 +1193,7 @@ export type BundleGcResponse = Message<"engram.app.v1.BundleGcResponse"> & { * Use `create(BundleGcResponseSchema)` to create a new message. */ export const BundleGcResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 32); + messageDesc(file_engram_app_v1_fleet, 31); /** * @generated from message engram.app.v1.SnapshotBlobGcRequest @@ -1251,7 +1218,7 @@ export type SnapshotBlobGcRequest = Message<"engram.app.v1.SnapshotBlobGcRequest * Use `create(SnapshotBlobGcRequestSchema)` to create a new message. */ export const SnapshotBlobGcRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 33); + messageDesc(file_engram_app_v1_fleet, 32); /** * Mirrors snapshot_blob_gc.rs SnapshotBlobSweepReport (ADR 0028 @@ -1314,7 +1281,7 @@ export type SnapshotBlobGcResponse = Message<"engram.app.v1.SnapshotBlobGcRespon * Use `create(SnapshotBlobGcResponseSchema)` to create a new message. */ export const SnapshotBlobGcResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 34); + messageDesc(file_engram_app_v1_fleet, 33); /** * ADR 0123 A3: the idempotent retirement request. Cordons the host with @@ -1350,7 +1317,7 @@ export type RetireHostRequest = Message<"engram.app.v1.RetireHostRequest"> & { * Use `create(RetireHostRequestSchema)` to create a new message. */ export const RetireHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 35); + messageDesc(file_engram_app_v1_fleet, 34); /** * @generated from message engram.app.v1.RetireHostResponse @@ -1379,7 +1346,7 @@ export type RetireHostResponse = Message<"engram.app.v1.RetireHostResponse"> & { * Use `create(RetireHostResponseSchema)` to create a new message. */ export const RetireHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 36); + messageDesc(file_engram_app_v1_fleet, 35); /** * @generated from message engram.app.v1.HostRetirement @@ -1413,7 +1380,7 @@ export type HostRetirement = Message<"engram.app.v1.HostRetirement"> & { * Use `create(HostRetirementSchema)` to create a new message. */ export const HostRetirementSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 37); + messageDesc(file_engram_app_v1_fleet, 36); /** * @generated from message engram.app.v1.RetirementBlocker @@ -1435,7 +1402,7 @@ export type RetirementBlocker = Message<"engram.app.v1.RetirementBlocker"> & { * Use `create(RetirementBlockerSchema)` to create a new message. */ export const RetirementBlockerSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 38); + messageDesc(file_engram_app_v1_fleet, 37); /** * Fleet, storage, and GC operations (ADR 0051 §2.3, rev 2026-06-10). @@ -1572,14 +1539,15 @@ export const FleetService: GenService<{ output: typeof FlushSessionResponseSchema; }, /** - * ADR 0018 async evacuation (POST /api/admin/sessions/:id/evacuate). + * ADR 0123 B: admit one teleport for a session (synchronous admission; + * the durable machine drives it). Replaces the ADR 0018 evacuate route. * - * @generated from rpc engram.app.v1.FleetService.EvacuateSession + * @generated from rpc engram.app.v1.FleetService.TeleportSession */ - evacuateSession: { + teleportSession: { methodKind: "unary"; - input: typeof EvacuateSessionRequestSchema; - output: typeof EvacuateSessionResponseSchema; + input: typeof TeleportSessionRequestSchema; + output: typeof TeleportSessionResponseSchema; }, /** * The three GC sweeps (ADR 0016 Phase C / ADR 0035 §5 / ADR 0028 diff --git a/crates/engram-coordinator/src/api/admin.rs b/crates/engram-coordinator/src/api/admin.rs index 2e283bf05..13efd4243 100644 --- a/crates/engram-coordinator/src/api/admin.rs +++ b/crates/engram-coordinator/src/api/admin.rs @@ -124,156 +124,14 @@ pub(crate) async fn flush_now_core( } } -// --------------------------------------------------------------------- -// ADR 0018 — session evacuation admin endpoints (async shape, commit 12) -// --------------------------------------------------------------------- -// -// Commit 12 rewrote evac from a synchronous -// snapshot→restore→rebind→harness-rebuild RPC into a state-machine -// transition + background scanner. The admin surface mirrors that -// split: -// -// - `POST /api/admin/sessions/:id/evacuate` — pause + flush + snapshot -// the source sandbox, mark the session `Evacuating`. Returns 202 -// immediately. The `evac_resumer` scanner picks the session up on -// its next tick (≤10s default) and drives it to Active on a peer. -// - `POST /api/admin/hosts/:id/cordon` / `uncordon` — flip the -// in-memory `HostState.draining` flag + PG `hosts.status` so the -// picker excludes the host. -// - `POST /api/admin/hosts/:id/drain` — cordon + fire Evacuating on -// every Active session on the host in parallel. Returns 202 with -// the list of session_ids being evacuated. -// -// The pause-before-flush ordering is what unblocks cross-host disk -// fidelity: the evict pipeline runs Pause → Flush → Snapshot -// → Destroy → transition_session, so the on-disk manifest the -// scanner restores from is bit-identical to what the source saw at -// pause time (no flush-vs-pause race; see ADR 0018 §"Commit 12 -// rework"). - -#[derive(Serialize)] -pub struct EvacuateSessionResponse { - pub session_id: SessionId, - /// "evacuating" — the session is paused, snapshotted, and the - /// `evac_resumer` scanner will resume it on a peer within the - /// next sweep interval (≤10s default). Operators can subscribe - /// to `GET /sessions/:id/events` to watch the - /// `Evacuating → Created → Active` chain land. - pub status: &'static str, -} - -/// `POST /api/admin/sessions/:id/evacuate` — mark the session -/// `Evacuating`. Pre: Active session with a bound sandbox. Post: the -/// source sandbox is paused, flushed, snapshotted, and destroyed; -/// PG row is at `Evacuating`; `evac_resumer` will resume on a peer. -/// -/// Returns 202 Accepted; the scanner is the actual deliverable. Use -/// the session events stream to observe the resume completing. -/// Transport-agnostic core for the evacuate primitive (ADR 0051). Marks -/// an Active session `Evacuating` via the shared eviction pipeline; the -/// `evac_resumer` scanner resumes it on a peer. The axum handler wraps -/// this in `(202, Json<_>)`; the gRPC `FleetService::evacuate_session` -/// reads `.session_id` + `.status` off the bare struct. -pub(crate) async fn evacuate_session_core( - state: &SharedState, - session_id: SessionId, -) -> Result { - let session = state.services.meta.get_session(session_id).await?; - if !matches!(session.status, engram_core::types::SessionState::Active) { - return Err(ApiError::Conflict(format!( - "evacuate only supported for Active sessions (got {})", - session.status.as_str(), - ))); - } - let Some(sandbox_id) = session.sandbox_id else { - return Err(ApiError::Conflict(format!( - "session {session_id} has no bound sandbox", - ))); - }; - - // Fire the shared eviction pipeline with `target_state = - // Evacuating` under an inline op claim (ADR 0079 — the claim is the - // per-session exclusion; the pipeline is the SAME evict-verb body, - // step-recorded, so a coordinator death mid-drive is re-driven by - // the executor's reclaim sweep from the recorded step). The only - // difference from an idle eviction is the terminal state, so both - // flows inherit the same recoverability invariants (snapshot durable - // in BlobStorage before destroy, PG state flips before host-side - // destroy). - let claim = crate::session_ops::OpClaim::try_acquire( - state, - session_id, - engram_core::types::session_op::OpKind::Evict, - serde_json::json!({ "target": "evacuating", "allow_park": false, "nominated": false }), - ) - .await - .map_err(|e| ApiError::Internal(format!("op claim acquire failed: {e}")))? - .ok_or_else(|| { - ApiError::Conflict(format!( - "session {session_id} is busy (an op is in flight); retry shortly", - )) - })?; - let _ = sandbox_id; // the pipeline re-reads the binding under the claim - let result = crate::idle_evictor::run_evict_pipeline( - &claim.as_ctx(), - engram_core::types::SessionState::Evacuating, - false, - false, - ) - .await; - match &result { - Ok(_) => { - claim - .finish(engram_core::types::session_op::OpState::Done, None) - .await - } - Err(e) => { - claim - .finish( - engram_core::types::session_op::OpState::Failed, - Some(&e.to_string()), - ) - .await - } - } - // Typed conversion (issue #1012): a wire-skewed source host mid-deploy - // must surface as a retryable 503, not an opaque 500. - result.map_err(ApiError::from)?; - - tracing::info!( - %session_id, - %sandbox_id, - "admin evacuate: session marked Evacuating; scanner will resume on peer", - ); - - Ok(EvacuateSessionResponse { - session_id, - status: "evacuating", - }) -} +// Administrative eviction and durable drain planning. #[derive(Serialize)] pub struct EvictIdleResponse { pub session_id: SessionId, - /// The evict op's observed outcome: "idle" (full suspend — paused, - /// flushed, snapshotted, sandbox destroyed, resume rebinds) or the - /// ADR 0074 rung-2 "evicting (parked-paused, rung 2)" (VM paused in - /// place, un-parked by the next prompt). A busy op lane / no-op - /// completion surfaces as a retryable Conflict instead. pub status: &'static str, } -/// Transport-agnostic core for the idle-eviction primitive (ADR 0051, -/// restored on the gRPC surface as `SessionService::EvictIdle`). The -/// explicit admin trigger for the idle-eviction pipeline: enqueues the -/// exact same evict verb (`idle_evictor::run_evict_pipeline`) the idle -/// detector's nomination enqueues on a timeout, -/// so it is a faithful stand-in for "the session went idle" — without waiting -/// out (or globally lowering) the idle TTL. Pre: Active session with a bound -/// sandbox. Post: session at `Idle`, memory snapshot durable in BlobStorage, -/// resumable. Synchronous (unlike `evacuate`, which hands off to the resumer -/// scanner): the pipeline runs inline and the session is `Idle` by the time -/// this returns. pub(crate) async fn evict_idle_core( state: &SharedState, session_id: SessionId, @@ -518,385 +376,65 @@ pub(crate) async fn uncordon_host_core( #[derive(Serialize)] pub struct DrainHostResponse { pub host_id: engram_core::HostId, - pub evacuating: Vec, - pub failures: Vec, + pub planned: Vec, + pub descended: Vec, + pub skipped: u32, } -#[derive(Debug, Serialize)] -pub struct DrainFailure { - pub session_id: SessionId, - pub error: String, +/// ADR 0123 A/B4: record the retirement request for `owner` and plan one +/// teleport per Active resident. The request is durable, so the teleport +/// scanner re-plans the host every tick: a resident that does not fit +/// today is tried again, and until it leaves it blocks the grant as a +/// visible `bound_sessions` blocker. `Conflict` when another owner holds +/// the cordon or the host is dead; idempotent on a retired host. +pub(crate) async fn request_host_retirement_core( + state: &SharedState, + host_id: engram_core::HostId, + owner: engram_core::types::host::CordonOwner, + reason: engram_core::types::teleport::TeleportReason, +) -> Result { + let meta = &state.services.meta; + let now = state.services.clock.now_utc(); + let requested = meta + .request_host_retirement(host_id, owner, reason.as_str(), now) + .await?; + if !requested { + let row = meta + .get_host(host_id) + .await? + .ok_or_else(|| ApiError::NotFound(format!("host {host_id} has no row")))?; + if row.status != engram_core::types::host::HostStatus::Retired { + let why = match row.cordon_owner { + Some(other) if other != owner => { + format!("host is cordoned by {}", other.as_str()) + } + _ => format!("host is {}", row.status.as_str()), + }; + return Err(ApiError::Conflict(why)); + } + } + Ok(crate::teleport::plan_host_teleports(state, host_id, reason).await) } -/// `POST /api/admin/hosts/:id/drain` — cordon the host, then run the -/// evict pipeline (target `Evacuating`) for every Active session on -/// it. Returns 202 with the per-session outcomes; the `evac_resumer` -/// scanner is responsible for completing each transition to Active -/// on a peer host. -/// -/// Concurrency: each session's eviction is independent and runs in -/// parallel — the source sandbox is paused on the source host -/// concurrently across sessions. The eviction lease serialises -/// per-session retries; two coord pods both running /drain on the -/// same host will see one win per session, the other no-op via the -/// lease guard. -/// Transport-agnostic core for the drain primitive (ADR 0051). Cordons -/// the host (durable `hosts.cordoned` bit), then fans out a live-first -/// move (snapshot-rehome fallback) for every Active session on it. The -/// axum `drain_host` handler wraps this in `(202, Json<_>)`; the gRPC -/// `FleetService::admin_drain_host` reads `.host_id`, `.evacuating`, and -/// `.failures` (each with `.session_id` + `.error`) off the bare struct. -/// -/// Issue #208: the per-session live-teleport verbs hold the session lease -/// across multi-second pause/capture/restore blackouts and a `JoinSet` -/// aborts in-flight tasks on drop — so the whole JoinSet is driven on its -/// own DETACHED task (not the caller's future). HTTP/gRPC cancellation -/// then only stops us OBSERVING; the per-session verbs still run to their -/// terminal commit/abort/parachute arms. This guard is preserved exactly. +/// `AdminDrainHost` is the retirement request with `owner = admin`. The +/// host is retired once it is empty; `UncordonHost{owner: admin}` cancels +/// the request before then. pub(crate) async fn admin_drain_host_core( state: &SharedState, host_id: engram_core::HostId, ) -> Result { - // ADR 0047: the durable cordon — heartbeats can't clobber it, every - // replica's picker reads it. A PG failure fails the drain (no - // in-memory fallback to half-drain behind). - match state - .services - .meta - .set_host_cordon( - host_id, - Some(engram_core::types::host::CordonOwner::Admin), - None, - ) - .await - { - Ok(()) => {} - Err(engram_core::MetaError::NotFound) => { - return Err(ApiError::NotFound(format!("host {host_id} has no row"))); - } - Err(e) => { - return Err(ApiError::Internal(format!( - "drain: set_host_cordon failed: {e}" - ))); - } - } - - // PG-authoritative list of sessions bound here (with budgets — ADR - // 0048 C8 needs them for the don't-strand guard). The in-memory - // `sandboxes_on_host` map is faster but can lag (post-restart - // rehydration window). For drain we use PG so a fresh coord pod - // can complete a drain initiated against a sibling. - let all_assignments = state - .services - .meta - .list_resident_assignments_with_budgets_on_host(host_id) - .await - .map_err(|e| ApiError::Internal(format!("drain: list sessions on host: {e}")))?; - - // Partition by residency flavor (2026-07-21 status-set audit - // finding 4 — a host holding only parked VMs used to "drain" - // successfully with an empty list and the roll operator stalled on - // `running_sandboxes == 0`): - // - Active → the live-first move below. - // - Parked → the descent evict via `descend_parked_session` - // (Parked → Evicting flip FIRST, then the nominated descent op — - // the reaper's contract; a raw non-nominated enqueue against a - // still-parked row is skipped by the pipeline's entry guard). - // The paused VM cannot be live-teleported, and - // Parked → Evacuating is not a legal edge. Once Idle it resumes - // anywhere on demand — same operator outcome as an evacuation. - // - Created / Unreachable / Evicting / Pending / Evacuating → - // their own machinery (boot, unreachable-recovery via prompt/ - // resume, the eviction pipeline, the queue scanner, the evac - // resumer) already converges them off a cordoned host; listing - // them here would double-drive those ops. - let mut assignments = Vec::new(); - let mut parked = Vec::new(); - let mut skipped = 0usize; - for a in all_assignments { - match a.status { - engram_core::types::SessionState::Active => assignments.push(a), - engram_core::types::SessionState::Parked => parked.push(a), - _ => skipped += 1, - } - } - - let mut parked_evacuating: Vec = Vec::new(); - let mut parked_failures: Vec = Vec::new(); - for a in &parked { - // Re-read the row: the descent needs `parked_at` (the dedup key) - // and the listing above is a snapshot — the session may have - // un-parked or died since. - let session = match state.services.meta.get_session(a.session_id).await { - Ok(s) => s, - Err(e) => { - parked_failures.push(DrainFailure { - session_id: a.session_id, - error: format!("drain: parked descent: get_session: {e}"), - }); - continue; - } - }; - match crate::idle_evictor::descend_parked_session(state, &session, "admin_drain").await { - Ok(true) => { - tracing::info!(%host_id, session_id = %a.session_id, sandbox_id = %a.sandbox_id, - "admin drain: parked session — descent initiated (Parked → Evicting → durable Idle)"); - parked_evacuating.push(a.session_id); - } - Ok(false) => { - // The Parked → Evicting flip raced (un-park ascent, - // delete, host death): the fresh status owns the session - // and this drain wave did NOT descend it. Surface it so - // the operator re-runs the drain rather than trusting a - // silently-shrunk work list. - parked_failures.push(DrainFailure { - session_id: a.session_id, - error: "drain: parked descent raced a concurrent transition; re-run the drain" - .to_string(), - }); - } - Err(e) => parked_failures.push(DrainFailure { - session_id: a.session_id, - error: format!("drain: parked descent: {e}"), - }), - } - } - - if assignments.is_empty() { - tracing::info!(%host_id, parked = parked.len(), other_resident = skipped, - "admin drain: host cordoned; no Active sessions to evacuate"); - return Ok(DrainHostResponse { - host_id, - evacuating: parked_evacuating, - failures: parked_failures, - }); - } - - // Fan out per-session moves. JoinSet so we collect outcomes - // without giving up on the first error. - // - // ADR 0045 C1: LIVE-FIRST. A drain is the teleport's marquee - // use-case — move each session losslessly to a peer; only fall - // back to the evict-to-Evacuating snapshot-rehome (the pre-C1 - // behavior, loses post-checkpoint state) when the live move - // can't run (flag off, no peer capacity, pre-C1 host) or fails - // back to the source. A Parachute failure already left the - // session Evacuating with the scanner armed — same end state as - // the fallback, so it counts as evacuating. - // - // Issue #208: the per-session bodies each run a live-teleport verb - // that holds the session lease across a multi-second pause/capture/ - // restore blackout. A `JoinSet` ABORTS all of its in-flight tasks - // when it is dropped — so if we held the JoinSet directly on this - // axum handler future, a client disconnect (drains run for minutes; - // LB timeouts and operator Ctrl-C are routine) would drop the - // handler, drop the JoinSet, and abort every in-flight migration - // mid-blackout. That strands frozen sources and forks sessions - // exactly like the inline teleport bug. Drive the whole JoinSet on - // its own detached task and merely await its JoinHandle here: HTTP - // cancellation then only stops us observing — the per-session verbs - // still run to their terminal arms. - let total = assignments.len(); - // Own a clone for the detached driver (the core borrows `state`; the - // spawned task needs a `'static` owned `SharedState`). - let driver_state: SharedState = (*state).clone(); - let driver = tokio::spawn(async move { - let state = driver_state; - let mut tasks = tokio::task::JoinSet::new(); - for a in &assignments { - let st = state.clone(); - let sid = a.session_id; - let mem_budget = a.mem_budget_mib; - let cpu_budget = a.cpu_budget_vcpus; - tasks.spawn(async move { - // ADR 0048 C8 don't-strand guard: before starting ANY move, - // confirm some SURVIVOR (a non-victim host) fits this session's - // budgets. If none does, do NOT begin the move — an Active - // session must never be parked Idle just because the fleet is - // full. Surface it as a failure so the operator aborts the wave. - let (repo, tag) = match st.services.meta.get_session(sid).await { - Ok(s) => { - let (r, t) = engram_core::types::session::split_image_ref(&s.image); - (r.to_string(), t.to_string()) - } - Err(e) => return (sid, Err(format!("get_session: {e}"))), - }; - // ADR 0068: a capacity-fit PREVIEW, not the move itself — - // no snapshot/manifest is loaded here, so no substrate - // requirement is derivable (or needed: the actual move, - // `evacuate_dead_source` or `migrate_session_live` below, - // re-derives `caps` from the real snapshot it restores and - // is the authoritative gate). The base capability gate - // (`host_meets_capabilities` with `Default` requirements) - // still applies through `placement_preview`. - let fit_ctx = crate::placement::ScheduleContext { - repo: &repo, - image_version: &tag, - snapshot_host: None, - memory_mib: Some(mem_budget.max(0) as u32), - cpu_budget_vcpus: Some(cpu_budget.max(0) as u32), - required_image_digest: None, - exclude_host: Some(host_id), - prefer_host: None, - caps: crate::placement::CapabilityRequirements::default(), - prefer_bundles: &[], - }; - match crate::placement::placement_preview( - st.services.meta.as_ref(), - &fit_ctx, - mem_budget, - cpu_budget as i64, - st.services.clock.now_utc(), - ) - .await - { - Ok(true) => {} - Ok(false) => { - return ( - sid, - Err("no surviving host has capacity for this session — \ - drain would strand it (scale up, then retry)" - .to_string()), - ); - } - Err(e) => return (sid, Err(format!("placement_preview: {e:?}"))), - } - - if crate::live_migration::live_teleport_enabled() { - let ctx = crate::placement::ScheduleContext { - repo: &repo, - image_version: &tag, - snapshot_host: None, - memory_mib: Some(mem_budget.max(0) as u32), - cpu_budget_vcpus: Some(cpu_budget.max(0) as u32), - required_image_digest: None, - exclude_host: Some(host_id), - prefer_host: None, - // ADR 0068: same preview posture as `fit_ctx` above — - // `migrate_session_live`'s own capture path is the - // authoritative gate for the live-teleport target. - caps: crate::placement::CapabilityRequirements::default(), - prefer_bundles: &[], - }; - let target = crate::placement::pick_for_session( - st.services.meta.as_ref(), - &st.host_registry, - &ctx, - st.services.clock.now_utc(), - ) - .await - .ok() - .map(|(h, _)| h); - if let Some(target) = target { - match crate::live_migration::migrate_session_live(&st, sid, target).await { - Ok(()) => return (sid, Ok(())), - Err(crate::live_migration::MigrateError::Parachute(e)) => { - tracing::warn!( - %sid, %host_id, error = %e, - "drain: live move parachuted; scanner rehome armed", - ); - return (sid, Ok(())); - } - Err(e) => { - tracing::warn!( - %sid, %host_id, error = %e, - "drain: live move failed; falling back to snapshot-rehome", - ); - } - } - } - } - // ADR 0079: inline op claim + the shared evict-verb - // pipeline (target Evacuating). A busy op lane (a - // concurrent eviction/resume owns the session) is not a - // drain failure: the session is being moved / handled by - // the other actor — fold to success, like `Skipped`. - let claim = match crate::session_ops::OpClaim::try_acquire( - &st, - sid, - engram_core::types::session_op::OpKind::Evict, - serde_json::json!({ - "target": "evacuating", "allow_park": false, "nominated": false - }), - ) - .await - { - Ok(Some(c)) => c, - Ok(None) => return (sid, Ok(())), - Err(e) => return (sid, Err(format!("op claim acquire: {e}"))), - }; - let outcome = crate::idle_evictor::run_evict_pipeline( - &claim.as_ctx(), - engram_core::types::SessionState::Evacuating, - false, - false, - ) - .await; - match &outcome { - Ok(_) => { - claim - .finish(engram_core::types::session_op::OpState::Done, None) - .await - } - Err(e) => { - claim - .finish( - engram_core::types::session_op::OpState::Failed, - Some(&e.to_string()), - ) - .await - } - } - // A `Skipped` (the session was already relocated, etc.) - // is folded to success — drain doesn't pin a destination, - // so unlike teleport (issue #214) there is no stale pin - // to unwind. Errors still surface as failures. - (sid, outcome.map(|_| ()).map_err(|e| e.to_string())) - }); - } - - let mut evacuating: Vec = Vec::new(); - let mut failures: Vec = Vec::new(); - while let Some(join) = tasks.join_next().await { - match join { - Ok((sid, Ok(()))) => evacuating.push(sid), - Ok((sid, Err(e))) => { - tracing::warn!(%sid, %host_id, error = %e, "drain: per-session evict/guard failed"); - failures.push(DrainFailure { - session_id: sid, - error: e, - }); - } - Err(e) => { - tracing::warn!(%host_id, error = %e, "drain: join error in per-session task"); - } - } - } - (evacuating, failures) - }); - - // Await the detached driver for the normal (connected) response. If - // the HTTP request is cancelled, this future is dropped — but the - // spawned `driver` (and therefore its JoinSet of per-session verbs) - // keeps running to completion, so no migration is aborted mid-move. - let (mut evacuating, mut failures) = driver.await.map_err(|join_err| { - ApiError::Internal(format!("drain: driver task panicked: {join_err}")) - })?; - evacuating.extend(parked_evacuating); - failures.extend(parked_failures); - - tracing::info!( - %host_id, - evacuating = evacuating.len(), - failures = failures.len(), - total, - "admin drain: per-session evac pipeline dispatched; scanner will resume each on a peer", - ); - + let plan = request_host_retirement_core( + state, + host_id, + engram_core::types::host::CordonOwner::Admin, + engram_core::types::teleport::TeleportReason::AdminDrain, + ) + .await?; Ok(DrainHostResponse { host_id, - evacuating, - failures, + planned: plan.planned, + descended: plan.descended, + skipped: plan.skipped, }) } diff --git a/crates/engram-coordinator/src/api/host_http.rs b/crates/engram-coordinator/src/api/host_http.rs index 3c05b9fae..951b867b4 100644 --- a/crates/engram-coordinator/src/api/host_http.rs +++ b/crates/engram-coordinator/src/api/host_http.rs @@ -1927,11 +1927,29 @@ pub async fn sandbox_ownership_core( // `owned=true` here kept those orphans alive (and their guest // memory pinned) indefinitely. Mirrors `session_owning_sandbox`'s // non-terminal predicate. - match state.services.meta.get_session(session_id).await { - Ok(s) => Ok(s.sandbox_id == Some(sandbox_id) && !s.status.is_terminal()), - Err(engram_core::MetaError::NotFound) => Ok(false), - Err(e) => Err(ApiError::Internal(format!("get_session: {e}"))), + // + // ADR 0123 B: an open teleport owns BOTH its endpoints. After commit + // the session names the destination, but the source is still the + // page server of a live move (or a held snapshot source awaiting + // release); answering `false` would let the host's export TTL sweep + // destroy it under a draining destination. + let session = match state.services.meta.get_session(session_id).await { + Ok(s) => s, + Err(engram_core::MetaError::NotFound) => return Ok(false), + Err(e) => return Err(ApiError::Internal(format!("get_session: {e}"))), + }; + if session.sandbox_id == Some(sandbox_id) && !session.status.is_terminal() { + return Ok(true); } + let moving = state + .services + .meta + .open_teleport_for_session(session_id) + .await + .map_err(|e| ApiError::Internal(format!("open_teleport_for_session: {e}")))?; + Ok(moving.is_some_and(|t| { + t.source_sandbox_id == sandbox_id || t.dest_sandbox_id == Some(sandbox_id) + })) } #[derive(serde::Serialize, serde::Deserialize)] diff --git a/crates/engram-coordinator/src/api/snapshot.rs b/crates/engram-coordinator/src/api/snapshot.rs index c6302d25d..3ae833ace 100644 --- a/crates/engram-coordinator/src/api/snapshot.rs +++ b/crates/engram-coordinator/src/api/snapshot.rs @@ -597,11 +597,7 @@ pub async fn ensure_active(state: &SharedState, id: SessionId) -> Result<(), Api // Evacuating session is NOT inline-resumable here: `resume_session` // has no Evacuating arm (it would 409), and `resume_from_idle`'s // rebind CAS only accepts Idle. Recovery is asynchronous — the - // `evac_resumer` scanner (ADR 0018 commit 12) relocates the session - // to the destination host and drives it `Evacuating → Created → - // Active`. Return a retryable 409 (like Queued/Pending) so the - // client polls /sessions/:id/events for the flip, rather than the - // misleading "only Idle / Created sessions can be resumed". + // teleport machine completes relocation. Return a retryable conflict. SessionState::Evacuating => Err(ApiError::Conflict( "session is relocating (operator drain / teleport); it will \ resume automatically — retry shortly." @@ -1043,32 +1039,10 @@ pub(crate) async fn resume_from_idle( // oracle orphan reap GCs it. The compensation (and its blind destroy // of a possibly-healthy VM) is unrepresentable under step-resume. - // ADR 0090 (#896 review, HIGH): a budget-exhausted evacuation lands - // here Idle WITH its unconfirmed source binding still in place — - // ownership of a maybe-live sandbox is never released on a guess. - // The guarded bind below requires an unbound row, so run the SAME - // teardown-confirmation gate the evac resumer uses: re-issue the - // idempotent destroy, probe, and only a confirmed-gone source - // authorizes the fenced clear. Unconfirmable → 503-retryable (the - // dead-host lane clears the binding once the source host is declared - // dead; a healthy-but-lagging teardown confirms on a later attempt). - // Checked BEFORE the restore so we never create a VM we may have to - // abandon. Normal idle-evicted rows are unbound and skip this leg, - // as does a step-resume re-entry after the clear committed. + // A retained Idle binding must be destroyed before a replacement is created. let mut session = session; if let (Some(source_host), Some(source_sandbox)) = (session.host_id, session.sandbox_id) { - crate::evac_resumer::confirm_source_teardown( - state, - source_host, - source_sandbox, - ctx.fence(), - ) - .await - .map_err(|e| { - ApiError::Unavailable(format!( - "resume: retained source binding could not be confirmed torn down: {e}" - )) - })?; + destroy_retained_sandbox(state, source_host, source_sandbox, ctx.fence()).await?; match state .services .meta @@ -1145,6 +1119,30 @@ pub(crate) async fn resume_from_idle( )) } +/// Destroy and probe a retained binding before a cold boot can replace it. +pub(crate) async fn destroy_retained_sandbox( + state: &SharedState, + host: engram_core::HostId, + sandbox: SandboxId, + fence: engram_core::traits::SessionFence, +) -> Result<(), ApiError> { + let backend = state.host_registry.backend_for(host).await?; + match backend.destroy(sandbox, fence).await { + Ok(()) => {} + Err(engram_core::SandboxError::NotFound) => return Ok(()), + Err(e) => return Err(e.into()), + } + match backend.probe_sandbox(sandbox).await { + Ok(probe) if !probe.known_to_backend && !probe.process_alive => Ok(()), + Err(engram_core::SandboxError::NotFound | engram_core::SandboxError::Unsupported(_)) => { + Ok(()) + } + other => Err(ApiError::Unavailable(format!( + "retained sandbox teardown not confirmed: {other:?}" + ))), + } +} + /// ADR 0028 Fix B — manual-resume flavor of the disk-only cold boot: /// fresh kernel boot mounting the session's `live_disk_manifest` on /// whichever host can take it, fresh harness. On-disk work survives; @@ -1154,91 +1152,62 @@ async fn resume_disk_only_cold_boot( ctx: &crate::session_ops::OpCtx<'_>, session: Session, ) -> Result { - use crate::evacuation::{evacuate_dead_source, EvacError}; - let state = ctx.state; let id = session.id; - - // The previous host isn't dead here (Idle = the sandbox was - // destroyed); clearing host_id disables `exclude_host` so a - // single-host deployment can recover onto itself. - let mut relocatable = session.clone(); - let origin = relocatable.host_id.take(); - - // A fresh VM comes up inside `evacuate_dead_source` — record the - // restore step first (crash-resume boundary). if !ctx.step("restore").await { return Err(fenced_error()); } - let receipt = evacuate_dead_source( + let mut spec = crate::boot_materializer::materialize_cold_boot(state, &session) + .await? + .ok_or_else(|| ApiError::Conflict("disk-only recovery requires an enabled image".into()))?; + spec.rootfs_manifest = session.live_disk_manifest; + let (repo, tag) = engram_core::types::session::split_image_ref(&session.image); + let context = crate::placement::ScheduleContext { + repo, + image_version: tag, + snapshot_host: None, + memory_mib: None, + cpu_budget_vcpus: None, + required_image_digest: None, + exclude_host: None, + prefer_host: session.host_id, + caps: Default::default(), + prefer_bundles: &[], + }; + let (host, backend) = crate::placement::pick_for_session( + state.services.meta.as_ref(), &state.host_registry, - &state.services.meta, - relocatable, - None, - // ADR 0116: lazily materialized — this path is disk-only by - // construction (no snapshot above), so the rung-2 branch always - // awaits it; the laziness keeps ONE call shape with the - // evac-resumer leg, where rung-1 must never run these reads. - crate::boot_materializer::materialize_cold_boot(state, &session), - None, - // ADR 0045 C2 (E2B fold, origin affinity): prefer the host the - // session last ran on — its NBD chunk cache (and base shm) are - // warm there. Soft tier-2: loses to capacity/draining, so this - // never strands the resume. - origin, - ctx.fence(), - // #800: `None` keeps this user-initiated single /resume path on its - // pre-#800 capacity-soft placement. The reserved (queue-on-no-fit) - // bound is wired on the drain-driven EVAC-SCANNER leg (`evac_resumer` - // — the #800 over-reservation wave); the resume verb's own - // MEMORY-snapshot path already queues via `placement_preview` - // (#795). Widening the reserved bound to this disk-only resume arm - // is a separate follow-up, out of #800's scope. - None, + &context, state.services.clock.now_utc(), ) .await - .map_err(|e| match &e { - EvacError::NoTargetAvailable(_) => ApiError::Unavailable(format!( - "no host can take the disk-only cold-boot recovery right now: {e}. \ - Retry shortly.", - )), - // Image un-enabled: structural, user-actionable — same message the - // eager pre-check used to produce. - EvacError::ColdBootUnavailable(_) => ApiError::Conflict(format!( - "session {id} has only a live disk manifest and its image `{}` is no \ - longer enabled — re-enable it (POST /api/enabled-images), then retry /resume", - session.image, - )), - // Transient slot-resolution read: retryable, not terminal. - EvacError::SpecResolution(_) => ApiError::Unavailable(format!( - "cold-boot spec resolution hit a transient error: {e}. Retry shortly.", - )), - _ => ApiError::Internal(format!("disk-only cold-boot recovery failed: {e}")), - })?; - - tracing::info!( - session_id = %id, - new_host = %receipt.new_host_id, - new_sandbox = %receipt.new_sandbox_id, - loss = receipt.loss.as_str(), - "resume: disk-only cold boot relocated session to Created — finishing harness rebuild", - ); - let _ = state - .emit( - id, - SessionEvent::StatusChanged { - from: SessionState::Idle, - to: SessionState::Created, - at: state.services.clock.now_utc(), - }, - ) - .await; + .map_err(|e| ApiError::Unavailable(format!("disk-only recovery placement: {e:?}")))?; + let sandbox = backend.create(spec).await?; + let binding_epoch = state + .services + .meta + .fenced_assign_sandbox(id, ctx.epoch, Some(sandbox), Some(host)) + .await? + .ok_or_else(fenced_error)?; + state.host_registry.record_sandbox_owner(sandbox, host); + crate::session_ops::transition_with_fence_emitting( + state, + id, + ctx.fence(), + SessionState::Created, + BindingDisposition::Retain, + vec![SessionEvent::StatusChanged { + from: session.status, + to: SessionState::Created, + at: state.services.clock.now_utc(), + }], + ) + .await?; if !ctx.step("bind").await { return Err(fenced_error()); } - bind_harness_generation(state, id, receipt.new_sandbox_id, receipt.binding_epoch).await?; + bind_harness_generation(state, id, sandbox, binding_epoch).await?; if !ctx.step("finish").await { return Err(fenced_error()); } @@ -1246,14 +1215,8 @@ async fn resume_disk_only_cold_boot( let outcome = finish_resume_to_active( state, &refreshed, - receipt.new_sandbox_id, - materialize_snapshot_resume( - state, - &refreshed, - receipt.new_sandbox_id, - receipt.binding_epoch, - ) - .await?, + sandbox, + materialize_snapshot_resume(state, &refreshed, sandbox, binding_epoch).await?, ctx.fence(), ) .await?; @@ -1427,14 +1390,8 @@ pub enum FinishResumeOutcome { /// transcript as-is rather than blocking the resume — the guest is /// already coherent; the worst case is a confusing-but-intact log. /// -/// Shared by [`resume_from_fc_snapshot`] (manual `/resume`) and -/// `evac_resumer::run_resume_pipeline` (operator drain / teleport) — the -/// two rung-1 entry points. The `cause` distinguishes them for the web -/// copy (ADR 0045 F1): the manual-`/resume` path resumes a session that -/// was idled after its host died, so it carries -/// [`RecoveryCause::HostFailureRecovery`]; the evac-resumer path is an -/// operator-initiated relocation, so it carries -/// [`RecoveryCause::PlannedRelocation`]. +/// Checkpoint recovery rewinds guest-derived events to its capture cursor. +/// Planned teleport does not rewind the transcript. pub async fn apply_rung1_rewind( state: &SharedState, session_id: SessionId, @@ -1512,10 +1469,6 @@ pub async fn apply_rung1_rewind( /// fresh sandbox bound: /// /// - [`resume_from_fc_snapshot`] (user-initiated `/resume` from Idle). -/// - [`crate::api::admin::evacuate_session`] (operator drain via the -/// admin endpoint). -/// - [`crate::evac_resumer`] scanner (drives `Evacuating → Created` -/// for sessions an operator drain marked; ADR 0044 K3). /// - [`resume_from_created`] dispatcher arm (manual recovery of a /// drained session). /// @@ -2199,9 +2152,7 @@ async fn bind_resumed_session( /// replica's `/exec` / `/shell` / `/prompt` resolves the new sandbox by /// reading that row ([`AppState::resolve_sandbox`]). /// -/// Shared with `bind_resumed_session` (the /resume path); exposed -/// `pub(crate)` so the admin evac endpoint and the `evac_resumer` -/// scanner (driving operator-drained sessions) can fire the same shape. +/// Shared by resume, exec, and teleport attachment. pub(crate) async fn bind_harness_generation( state: &SharedState, id: SessionId, @@ -2913,7 +2864,7 @@ mod evicting_gate_tests { /// ADR 0018: an `ensure_active` (/exec, /events) landing while an /// operator drain / teleport has the session at `Evacuating` must - /// return a RETRYABLE 409 — the `evac_resumer` relocates it back to + /// return a RETRYABLE 409 — the `teleport` relocates it back to /// Active asynchronously. It must NOT route to `resume_session` (which /// has no Evacuating arm and would 409 with the misleading "only Idle / /// Created can be resumed"), and must leave the session untouched. diff --git a/crates/engram-coordinator/src/checkpoint_retention.rs b/crates/engram-coordinator/src/checkpoint_retention.rs index 62b84d4a2..e829e1a10 100644 --- a/crates/engram-coordinator/src/checkpoint_retention.rs +++ b/crates/engram-coordinator/src/checkpoint_retention.rs @@ -51,7 +51,7 @@ impl Default for CheckpointRetentionConfig { } } -/// Spawn the sweeper. Mirrors [`crate::evac_resumer::spawn`]. +/// Spawn the sweeper. Mirrors [`crate::teleport::spawn`]. pub fn spawn(cfg: CheckpointRetentionConfig, state: SharedState) -> tokio::task::JoinHandle<()> { tokio::spawn(async move { let mut tick = tokio::time::interval(cfg.poll_interval); diff --git a/crates/engram-coordinator/src/dead_host.rs b/crates/engram-coordinator/src/dead_host.rs index d40a2cce8..364ee03a8 100644 --- a/crates/engram-coordinator/src/dead_host.rs +++ b/crates/engram-coordinator/src/dead_host.rs @@ -67,7 +67,7 @@ //! `Evacuating` — the reactive auto-evac was the documented bug //! source (the resume-from-idle wedge, the deploy-storm cascade), so //! `Evacuating` is now reached *only* via operator drain (ADR 0044 -//! K3), and the `evac_resumer` scanner relocates only those. +//! K3), and the `teleport` scanner relocates only those. use std::sync::Arc; use std::time::Duration; diff --git a/crates/engram-coordinator/src/enable_scanner.rs b/crates/engram-coordinator/src/enable_scanner.rs index b5f144bd9..59cd89d2a 100644 --- a/crates/engram-coordinator/src/enable_scanner.rs +++ b/crates/engram-coordinator/src/enable_scanner.rs @@ -2,7 +2,7 @@ //! //! Background task that drives `enable_jobs` rows through //! `pending → materializing → capturing → prestaging → ready | failed`. -//! Sibling to [`crate::evac_resumer`]: same polling shape, same +//! Sibling to [`crate::teleport`]: same polling shape, same //! shared-state surface, distinct table. //! //! ## Flow @@ -199,7 +199,7 @@ impl EnableScannerConfig { /// Spawn the scanner as a background task. Caller holds the /// JoinHandle for the process lifetime; dropping aborts the loop. -/// Mirrors [`crate::evac_resumer::spawn`]. +/// Mirrors [`crate::teleport::spawn`]. pub fn spawn(cfg: EnableScannerConfig, state: SharedState) -> tokio::task::JoinHandle<()> { tokio::spawn(async move { let mut tick = tokio::time::interval(cfg.poll_interval); diff --git a/crates/engram-coordinator/src/evac_resumer.rs b/crates/engram-coordinator/src/evac_resumer.rs deleted file mode 100644 index 53ef6cdd4..000000000 --- a/crates/engram-coordinator/src/evac_resumer.rs +++ /dev/null @@ -1,928 +0,0 @@ -//! ADR 0018 commit 12c — Evacuating-session resumer. -//! -//! Background task that turns `Evacuating` sessions back into `Active` -//! sessions on a peer host. Sibling to [`crate::dead_host`]: same -//! polling shape, same shared-state surface, distinct entry point on -//! the state machine. -//! -//! ## Flow -//! -//! 1. Tick: read `sessions WHERE status = 'evacuating'` (with -//! `evac_attempts`) via -//! [`MetadataStore::list_evacuating_sessions`]. -//! 2. For each candidate, if `evac_attempts >= max_attempts` → -//! `Evacuating → Idle` and stop trying (user can `/resume`). -//! 3. Otherwise bump `evac_attempts` atomically, then run the -//! relocation pipeline: -//! - [`crate::evacuation::evacuate_dead_source`] picks a peer host, -//! restores from the session's `live_disk_manifest` and/or latest -//! snapshot, rebinds PG `(host_id, sandbox_id)`, transitions -//! `Evacuating → Created`. -//! - [`crate::api::snapshot::bind_harness_generation`] registers the -//! session→sandbox map on the target host-agent (the coordinator -//! keeps no in-memory binding — `sessions.sandbox_id` is the -//! authority, ADR 0047). -//! - [`crate::api::snapshot::finish_resume_to_active`] runs the -//! harness rebuild + drives `Created → Active`. -//! 4. On any error in the pipeline, the session is left at its current -//! state — Evacuating (retry next tick) or Created (a later scanner -//! tick re-picks it up via the operator-/exec-driven `/resume` -//! path). The pre-bump idempotency lives in -//! `evacuate_dead_source` (PG rebind is `assign_*` which tolerates -//! re-runs) and in `bind_harness_generation` (an idempotent host RPC). -//! -//! ## Why this pattern -//! -//! Per `[async_via_state_machine]` — drain is a multi-host, multi-step -//! operation. Synchronous orchestration would couple the source -//! handler to a known target and conflate retry domains; the -//! state-machine+scanner shape decouples them. Source writes -//! "session is ready to be continued"; scanner finds a healthy peer. -//! Operator-initiated drain (ADR 0044 K3) is the sole producer of -//! `Evacuating` — ADR 0045 Phase A retired the reactive dead-host / -//! NBD-loss producers (the dead-host detector now routes recoverable -//! sessions to `Idle` for lazy `/resume`). -//! -//! The scanner is single-coord-pod safe because each per-session -//! advance starts with `bump_evac_attempts` (atomic +1) followed by -//! `evacuate_dead_source`'s pick + restore + rebind. Two coord pods -//! racing on the same session would both observe `Evacuating`, both -//! attempt restore, and the second's `transition_session(Created)` -//! would see the row already at `Created` and surface `Conflict`. -//! That's a harmless duplicate sandbox on the target (cleaned up by -//! orphan reap) — same shape as the dead-host detector's existing -//! advisory-lock race. Tightening with an advisory lock per session -//! is a follow-up if duplicate-restore counts ever rise above zero. - -use std::time::Duration; - -use chrono::Utc; -use engram_core::types::BindingDisposition; -use engram_core::types::{Session, SessionState}; - -use crate::api::snapshot::{finish_resume_to_active, FinishResumeOutcome}; -use crate::evacuation::{evacuate_dead_source, EvacError}; -use crate::state::{SessionEvent, SharedState}; - -/// Issue #214: max age of an operator teleport pin before the evac -/// scanner treats it as a leak and ignores + clears it. A pin is meant -/// to be consumed within one scanner tick (~seconds) of being set; one -/// that survives 15 minutes can only have leaked from a path that set it -/// without driving the session through `Evacuating`. Defense in depth -/// behind the core teleport_session fix — degrades any future leak to -/// default placement instead of a strict hijack. -const TELEPORT_PIN_TTL: chrono::Duration = chrono::Duration::minutes(15); - -/// Issue #214: is a teleport pin stamped at `set_at` old enough (as of -/// `now`) to be treated as a leak? A `None` stamp (pin set before the -/// 0065 migration) is never aged out — we keep honoring legacy pins. -fn teleport_pin_aged( - set_at: Option>, - now: chrono::DateTime, -) -> bool { - set_at - .map(|t| now.signed_duration_since(t) > TELEPORT_PIN_TTL) - .unwrap_or(false) -} - -#[derive(Clone, Debug)] -pub struct EvacResumerConfig { - /// How often to sweep for Evacuating sessions. The scanner picks - /// up new entries from operator drains (ADR 0044 K3) — the sole - /// producer of `Evacuating` since ADR 0045 Phase A retired the - /// reactive triggers. Default 10s matches - /// `DeadHostConfig::poll_interval` so the two scanners share the - /// same operational cadence. - pub poll_interval: Duration, - /// Retry budget per session before falling back to `Idle`. At the - /// default 10s cadence, 20 attempts is ~3 minutes — long enough - /// to ride out a transient capacity / image-prefetch shortfall on - /// peer hosts during a rolling restart, short enough that a truly - /// stuck session surfaces to the user as `Idle` (manual /resume) - /// before they assume it's gone. - pub max_attempts: u32, -} - -impl Default for EvacResumerConfig { - fn default() -> Self { - Self { - poll_interval: Duration::from_secs(10), - max_attempts: 20, - } - } -} - -/// Spawn the resumer as a background task. Caller holds the JoinHandle -/// for the process lifetime; dropping aborts the loop. Mirrors -/// [`crate::dead_host::spawn`]. -pub fn spawn(cfg: EvacResumerConfig, state: SharedState) -> tokio::task::JoinHandle<()> { - tokio::spawn(async move { - let mut tick = tokio::time::interval(cfg.poll_interval); - // Skip the first immediate tick — coord just started, give - // hosts a beat to heartbeat in before we pick. - tick.tick().await; - loop { - tick.tick().await; - if let Err(e) = run_once(&cfg, &state).await { - tracing::warn!(error = %e, "evac-resumer tick failed; will retry"); - } - } - }) -} - -/// Single scanner tick. `pub(crate)` so live-PG tests can drive the -/// scanner deterministically without `tokio::spawn`-ing the loop. -/// Production code uses [`spawn`] which calls this on a timer. -pub async fn run_once( - cfg: &EvacResumerConfig, - state: &SharedState, -) -> Result<(), Box> { - let candidates = state.services.meta.list_evacuating_sessions().await?; - if candidates.is_empty() { - return Ok(()); - } - tracing::debug!( - count = candidates.len(), - "evac-resumer found Evacuating sessions" - ); - for (session, attempts) in candidates { - if let Err(e) = advance_one(cfg, state, session, attempts).await { - // Keep going — one wedged session shouldn't stall the - // sweep. The per-session log already carries `error = - // %e`; this is the loop-level swallow. - tracing::warn!(error = %e, "evac-resumer per-session advance failed"); - } - } - Ok(()) -} - -// ADR 0019 / telemetry restoration (#526): scanner-driven work has no -// request span to inherit — an explicit root (carrying `session_id`) so -// the relocation pipeline's spans correlate instead of exporting as -// disconnected roots. -#[tracing::instrument(name = "evac_resumer.advance_one", skip_all, fields(session_id = %session.id))] -async fn advance_one( - cfg: &EvacResumerConfig, - state: &SharedState, - session: Session, - attempts: u32, -) -> Result<(), Box> { - let session_id = session.id; - // ADR 0045 C1 / ADR 0079: claim the session's op lane before driving - // a resume — a live migration's inline claim (or a peer pod's op) may - // be mid-flight on this session; without the claim two actors can - // double-restore. Claim-or-give-up: a busy lane means the holder owns - // the session; we re-scan next tick. The claim rides the op log - // (kind = resume, evac flavor); if this pod dies mid-pipeline the - // reclaim sweep re-claims the row and the resume verb terminally - // fails it (status Evacuating is not verb-resumable), freeing the - // lane for the next tick's fresh claim. - let Some(claim) = crate::session_ops::OpClaim::try_acquire( - state, - session_id, - engram_core::types::session_op::OpKind::Resume, - serde_json::json!({ "flavor": "evac" }), - ) - .await - .map_err(|e| format!("evac-resumer op claim acquire: {e}"))? - else { - tracing::debug!(%session_id, "evac-resumer: an op owns the session; skipping this tick"); - return Ok(()); - }; - let result = advance_one_claimed(cfg, state, session, attempts, &claim).await; - match &result { - Ok(()) => { - claim - .finish(engram_core::types::session_op::OpState::Done, None) - .await - } - Err(e) => { - claim - .finish( - engram_core::types::session_op::OpState::Failed, - Some(&e.to_string()), - ) - .await - } - } - result -} - -/// The claimed body of [`advance_one`] — runs with the op lane held. -async fn advance_one_claimed( - cfg: &EvacResumerConfig, - state: &SharedState, - session: Session, - attempts: u32, - claim: &crate::session_ops::OpClaim, -) -> Result<(), Box> { - let session_id = session.id; - // Issue #211 (ADR 0044 K5 shape, copied from idle_evictor): the - // `candidates` list is a sweep snapshot up to ~10s stale. Without a - // post-claim re-read we could drive a resume — restoring a live VM - // on a peer host and binding it — onto a row that has since gone - // terminal (a `DELETE /sessions/:id` flips Evacuating→Failed) or was - // already relocated by a competitor. The op claim is held until the - // pipeline finishes, so once we hold it any concurrent actor has - // fully completed; re-read the authoritative PG state and skip - // unless the row is still `Evacuating` (the only legal input to the - // resume pipeline). This keeps a stale tick from binding a fresh - // sandbox onto a terminal row (defeating the orphan reap) — the same - // failure the guarded binds in `bind_resumed_session` reject, caught - // earlier so we never create the VM in the first place. - let mut session = match state.services.meta.get_session(session_id).await { - Ok(s) if s.status != SessionState::Evacuating => { - tracing::info!( - %session_id, - state = s.status.as_str(), - "evac-resumer: session no longer Evacuating after claim (terminated or \ - relocated by a peer) — skipping", - ); - return Ok(()); - } - Ok(s) => s, - Err(e) => { - return Err(format!("evac-resumer: re-read session state after claim: {e}").into()); - } - }; - - // Retry budget exhausted → fall back to Idle so the user can - // `/resume` manually. Idle is a legal target from Evacuating per - // the legality table; the row's snapshot lineage is already - // durable (it was captured before the pipeline marked the - // session Evacuating in the evict pipeline), so /resume - // from Idle restores cleanly. Checked BEFORE the teardown - // confirmation so a source that stays connected but can never - // positively confirm (its failures burn this same budget below) - // reaches this terminal instead of wedging Evacuating forever. - // The fallback may then leave the source binding IN PLACE — that - // is deliberate: ownership of an unconfirmed sandbox is never - // released here. The `/resume` this hands off to runs the SAME - // confirmation gate (`resume_from_idle`'s stale-binding leg) and - // performs the fenced clear itself once the source is confirmed - // gone — or fails 503-retryable until the dead-host lane clears - // the binding. Ownership release stays behind one gate. - if attempts >= cfg.max_attempts { - match state - .services - .meta - .transition_session(session_id, SessionState::Idle, BindingDisposition::Retain) - .await - { - Ok(prev) => { - tracing::warn!( - %session_id, - attempts, - max_attempts = cfg.max_attempts, - "evac-resumer budget exhausted; session left at Idle for user /resume", - ); - let _ = state - .emit( - session_id, - SessionEvent::StatusChanged { - from: prev, - to: SessionState::Idle, - at: state.services.clock.now_utc(), - }, - ) - .await; - } - Err(e) => { - tracing::warn!( - %session_id, - error = %e, - "evac-resumer fallback transition Evacuating→Idle failed", - ); - } - } - // Gave up relocating — drop any teleport pin so a later manual - // /resume isn't constrained to the (evidently unavailable) target. - let _ = state - .services - .meta - .set_teleport_target(session_id, None) - .await; - return Ok(()); - } - - // ADR 0090: `Evacuating` retains the outgoing binding until teardown - // is positively confirmed. The drain's evict op has already issued a - // best-effort destroy, but an acknowledged host verb can still have a - // durable, not-yet-applied effect. Restoring from the snapshot while - // that sandbox is alive would give this session two owners across the - // fleet. Re-issue the idempotent destroy through the source backend, - // then independently probe it. Only a negative probe (or host-side - // NotFound) authorizes the fenced binding clear below. - match (session.host_id, session.sandbox_id) { - (Some(source_host), Some(source_sandbox)) => { - if let Err(e) = - confirm_source_teardown(state, source_host, source_sandbox, claim.fence()).await - { - // Confirmation failures burn the SAME budget as pipeline - // failures — without this, a persistently unconfirmable - // teardown never reaches the exhaustion fallback above and - // the session wedges with no terminal state. - let _ = state.services.meta.bump_evac_attempts(session_id).await; - return Err(e); - } - match state - .services - .meta - .fenced_assign_sandbox( - session_id, - claim.fence().epoch as i64, - None, - Some(source_host), - ) - .await - { - Ok(_) => { - state.host_registry.invalidate_sandbox(source_sandbox); - session.sandbox_id = None; - } - Err(engram_core::MetaError::Conflict(_)) => { - crate::metrics::note_fenced_write(); - return Ok(()); - } - Err(e) => { - return Err( - format!("evac-resumer: clear confirmed-dead source binding: {e}").into(), - ); - } - } - } - (None, Some(source_sandbox)) => { - // Unconfirmable by construction — burn the budget so this - // (should-be-impossible) shape also reaches the exhaustion - // fallback instead of wedging. - let _ = state.services.meta.bump_evac_attempts(session_id).await; - return Err(format!( - "evac-resumer: session {session_id} retains source sandbox {source_sandbox} \ - without a source host; refusing to release ownership" - ) - .into()); - } - (_, None) => {} - } - - // Bump pre-pipeline. A pipeline failure leaves the counter - // incremented and the session at Evacuating — next tick retries - // until the budget runs out. Bumping post-success isn't needed - // because `transition_session(Evacuating)` resets the counter on - // every entry per migration 0037's CASE expression. - let new_attempts = state.services.meta.bump_evac_attempts(session_id).await?; - tracing::info!( - %session_id, - attempt = new_attempts, - max_attempts = cfg.max_attempts, - "evac-resumer: starting resume attempt", - ); - - run_resume_pipeline(state, session, claim.fence()).await?; - Ok(()) -} - -pub(crate) async fn confirm_source_teardown( - state: &SharedState, - source_host: engram_core::HostId, - source_sandbox: engram_core::SandboxId, - fence: engram_core::traits::SessionFence, -) -> Result<(), Box> { - let backend = state.host_registry.backend_of(source_host).ok_or_else(|| { - format!( - "evac-resumer: source host {source_host} is not connected; retaining ownership of \ - sandbox {source_sandbox}" - ) - })?; - - match backend.destroy(source_sandbox, fence).await { - Ok(()) => {} - Err(engram_core::SandboxError::NotFound) => return Ok(()), - Err(e) => { - return Err(format!( - "evac-resumer: source destroy for sandbox {source_sandbox} on host \ - {source_host} was not confirmed: {e}" - ) - .into()); - } - } - - match backend.probe_sandbox(source_sandbox).await { - Ok(probe) if !probe.known_to_backend && !probe.process_alive => Ok(()), - Ok(probe) => Err(format!( - "evac-resumer: source sandbox {source_sandbox} on host {source_host} still owns \ - the session after destroy acknowledgement (known={}, alive={}); retaining \ - coordinator ownership", - probe.known_to_backend, probe.process_alive, - ) - .into()), - Err(engram_core::SandboxError::NotFound) => Ok(()), - // An old host-agent mid-roll has no probe RPC (`Unimplemented` → - // `Unsupported`, per the HostClient trait doc). Proceed on the - // strength of the confirmed destroy above — the same "no probe - // available" posture `reconcile::flip_missing` documents. A drain - // is exactly when mixed agent versions exist; refusing here wedged - // every evacuation off a not-yet-rolled host. - Err(engram_core::SandboxError::Unsupported(_)) => Ok(()), - Err(e) => Err(format!( - "evac-resumer: source teardown probe for sandbox {source_sandbox} on host \ - {source_host} failed: {e}; retaining coordinator ownership" - ) - .into()), - } -} - -async fn run_resume_pipeline( - state: &SharedState, - session: Session, - fence: engram_core::traits::SessionFence, -) -> Result<(), Box> { - let session_id = session.id; - let snapshot = state - .services - .meta - .latest_snapshot_for_session(session_id) - .await?; - - // ADR 0028 A.log: capture the rung-1 rewind cursor before the - // snapshot moves into evacuate_dead_source. Only a coherent - // checkpoint (memory present) rewinds; the receipt's - // `EvacLoss::None` confirms rung-1 actually happened. - let rewind_cursor = snapshot - .as_ref() - .filter(|s| s.memory_manifest.is_some()) - .and_then(|s| s.events_cursor); - - // #800 (RESERVED evac placement): the session's reserved 2D budget, - // resolved from the enabled image the same way the resume verb resolves - // it. `None` (image un-enabled) keeps the pre-#800 capacity-soft - // placement inside `evacuate_dead_source`. Deliberately the LIGHT - // probe (ADR 0116): the full cold-boot materialization is passed - // below as an un-awaited future that only the disk-only rung runs — - // a rung-1 memory-snapshot recovery must never fail (or even do the - // reads) for slot resolution it will not consult. - let evac_budget = - crate::boot_materializer::resolve_resume_budget(&state.services.meta, &session).await; - - // ADR 0045 Phase F: an operator-pinned teleport destination, if any. - // Honored strictly (a bad pin retries then falls back to Idle, never - // silently lands elsewhere); cleared below once the session resolves. - // - // Issue #214 defense in depth: a pin is supposed to be consumed within - // seconds of being set (teleport marks the session Evacuating, the next - // scanner tick resolves it). A pin still present long after it was set - // is a LEAK — some path set it without the session ever entering - // Evacuating (the core fix closes the known teleport no-op path; this - // backstops any future one). Strictly honoring a stale pin would hijack - // this evacuation onto a possibly full/gone host, then burn the budget - // to a forced-Idle strand. Instead: ignore + clear + warn on an aged - // pin so this evacuation degrades to default capacity-ranked placement. - let require_host = match state.services.meta.get_teleport_target(session_id).await { - Ok(Some((host, set_at))) => { - if teleport_pin_aged(set_at, state.services.clock.now_utc()) { - tracing::warn!( - %session_id, - stale_target = %host, - set_at = ?set_at, - ttl_secs = TELEPORT_PIN_TTL.num_seconds(), - "evac-resumer: teleport pin older than TTL — treating as a leaked pin; \ - ignoring + clearing it, falling back to default placement (issue #214)", - ); - let _ = state - .services - .meta - .set_teleport_target(session_id, None) - .await; - None - } else { - Some(host) - } - } - Ok(None) => None, - Err(e) => { - tracing::warn!(%session_id, error = %e, - "get_teleport_target failed; treating as unpinned"); - None - } - }; - - let receipt = match evacuate_dead_source( - &state.host_registry, - &state.services.meta, - session.clone(), - snapshot, - // ADR 0116: un-awaited — only the disk-only rung inside runs it. - crate::boot_materializer::materialize_cold_boot(state, &session), - require_host, - // No origin preference: every scanner producer (drain, dead - // host, migration parachute) is moving AWAY from the source. - None, - fence, - evac_budget, - state.services.clock.now_utc(), - ) - .await - { - Ok(r) => r, - // ADR 0123 B8: keep the reservation until capacity is available. - // C2 replaces this driver with the durable teleport machine. - Err(EvacError::NoCapacityQueue) => { - tracing::info!(%session_id, "evac-resumer: no capacity; retry later"); - return Ok(()); - } - // ADR 0028 Fix B fail-fast: structural errors can never be - // fixed by retrying — the pre-Fix-B behavior of letting the - // budget loop burn 20 attempts (~3 min of RestoreFailed churn - // in the cf4d4afd incident) just delayed the honest terminal - // state. NoRecoverableState (nothing to restore from) → Dead; - // ColdBootUnavailable (disk exists, image gone) → Idle, so - // re-enabling the image + /resume can still recover the disk. - Err(e) if e.is_structural() => { - let target = match &e { - EvacError::NoRecoverableState => SessionState::Dead, - _ => SessionState::Idle, - }; - tracing::warn!( - %session_id, - error = %e, - target = %target.as_str(), - "evac-resumer: structural failure — failing fast instead of burning budget", - ); - match state - .services - .meta - .transition_session(session_id, target, BindingDisposition::RequireUnbound) - .await - { - Ok(prev) => { - let _ = state - .emit( - session_id, - SessionEvent::StatusChanged { - from: prev, - to: target, - at: state.services.clock.now_utc(), - }, - ) - .await; - } - Err(te) => { - tracing::warn!( - %session_id, - error = %te, - "evac-resumer: structural fail-fast transition failed", - ); - } - } - // Session left Evacuating terminally — drop any teleport pin. - let _ = state - .services - .meta - .set_teleport_target(session_id, None) - .await; - return Ok(()); - } - Err(e) => return Err(Box::new(e) as Box), - }; - - // Resolved onto a peer (Created) — the teleport pin is consumed. - let _ = state - .services - .meta - .set_teleport_target(session_id, None) - .await; - - tracing::info!( - %session_id, - new_host = %receipt.new_host_id, - new_sandbox = %receipt.new_sandbox_id, - loss = receipt.loss.as_str(), - "evac-resumer: rebound to peer at Created — finishing harness rebuild", - ); - - // StatusChanged{prev → Created} so SSE subscribers see the move. - // `evacuate_dead_source` already committed the transition to - // Created in PG, so we emit synthesized event with from=Evacuating. - let _ = state - .emit( - session_id, - SessionEvent::StatusChanged { - from: SessionState::Evacuating, - to: SessionState::Created, - at: state.services.clock.now_utc(), - }, - ) - .await; - - // ADR 0073: evac restore is a fresh-spawn generation — mint. - crate::api::snapshot::bind_harness_generation( - state, - session_id, - receipt.new_sandbox_id, - receipt.binding_epoch, - ) - .await?; - - // ADR 0028 A.log: warm rung-1 recovery — rewind the transcript to - // the checkpoint's cursor + emit the recovery boundary. Gated on - // EvacLoss::None (memory was actually restored); a rung-2 cold - // boot carries no cursor and skips this. No-op if the checkpoint - // was the head. - // ADR 0045 F1: this path is only reached via operator drain / - // teleport (Phase A retired the reactive dead-host producer), so the - // rewind is a planned relocation, not a host failure. - if receipt.loss == engram_core::types::evacuation::EvacLoss::None { - crate::api::snapshot::apply_rung1_rewind( - state, - session_id, - rewind_cursor, - crate::state::RecoveryCause::PlannedRelocation, - ) - .await; - } - - // Refresh the session row so finish_resume_to_active sees the - // freshly-bound host_id + sandbox_id. - let session_refreshed = state.services.meta.get_session(session_id).await?; - match finish_resume_to_active( - state, - &session_refreshed, - receipt.new_sandbox_id, - crate::boot_materializer::materialize_snapshot_resume( - state, - &session_refreshed, - receipt.new_sandbox_id, - receipt.binding_epoch, - ) - .await?, - fence, - ) - .await - { - Ok(FinishResumeOutcome::Active) => { - tracing::info!( - %session_id, - "evac-resumer: session reached Active on peer host", - ); - } - Ok(FinishResumeOutcome::CreatedHarnessFailed { message: e, .. }) => { - tracing::warn!( - %session_id, - error = %e, - "evac-resumer: harness rebuild failed on peer; session left at Created — \ - user /resume retries from there (ADR 0090: the resume op now owns \ - backoff + budget). Scanner won't re-pick (status != Evacuating).", - ); - } - Err(e) => { - tracing::warn!( - %session_id, - error = %e, - "evac-resumer: finish_resume_to_active errored; session left at Created", - ); - } - } - Ok(()) -} - -#[cfg(test)] -// tests drive a live system; wall clock/OS entropy here is input, not a decision source (ADR 0098 D1) -#[allow(clippy::disallowed_methods)] -mod tests { - // The scanner's full loop exercises Postgres + HostRegistry + - // finish_resume_to_active; the per-step plumbing is unit-tested - // via the existing evacuation / api::snapshot tests. End-to-end - // coverage lives in: - // - `admin_evac_live_pg` (commit 12i): operator-drain triggers - // Evacuating, scanner picks it up, session reaches Active on - // peer. - // - `evac_resumer_budget_falls_back_to_idle` (commit 12i): - // simulate 20 failed bumps, observe Evacuating → Idle - // fallback. - // - dev-vm integration-evac-test.sh: real two-host drain. - // - // ADR 0028 Fix B adds the structural fail-fast tests below: a - // structurally-unrecoverable session must reach its terminal - // state on the FIRST attempt, not after burning the 20-attempt - // budget (~3 min of churn in the cf4d4afd incident). - - use super::*; - use crate::config::CoordinatorConfig; - use crate::host_registry::HostRegistry; - use crate::state::tests::MiniMeta; - use crate::state::AppState; - use crate::Services; - use engram_core::traits::MetadataStore; - use engram_core::types::session::SessionMode; - use std::sync::Arc; - - fn evacuating_session(live_disk: Option) -> Session { - Session { - id: engram_core::SessionId::new(), - status: SessionState::Evacuating, - host_id: None, - sandbox_id: None, - image: "test/repo:evac-test".into(), - mode: SessionMode::Agent, - created_at: Utc::now(), - last_active_at: Utc::now(), - last_event_at: None, - live_disk_manifest: live_disk, - park_rung: 0, - parked_at: None, - suggested_title: None, - } - } - - fn build_state(session: Session) -> (SharedState, Arc) { - let tmp = std::env::temp_dir().join(format!("evac-resumer-test-{}", session.id)); - std::fs::create_dir_all(&tmp).unwrap(); - let meta = Arc::new(MiniMeta::new(session)); - let host_registry = Arc::new(HostRegistry::new( - meta.clone() as Arc - )); - let services = Services { - meta: meta.clone(), - host: host_registry.clone() as Arc, - secrets: Arc::new(engram_secrets_dev::InMemorySecretStore::new()), - kek: Arc::new(engram_crypto::EnvVarKeyProvider::from_bytes( - [0u8; 32], "test:v1", - )), - oci: Arc::new(engram_oci::OciClient::new(Arc::new( - engram_oci::AnonymousResolver, - ))), - auth_resolver: Arc::new(engram_oci::AnonymousResolver), - blob: Arc::new(engram_storage_local::LocalBlobStorage::new( - tmp.join("blobs"), - )), - chunk_store: engram_chunk_store::ChunkStore::new(Arc::new( - engram_storage_local::LocalBlobStorage::new(tmp.join("blobs")), - )), - host_pool: Arc::new(engram_protocol::grpc_pool::GrpcHostPool::new()), - materialize_dir: None, - clock: Arc::new(engram_core::traits::SystemClock::new()), - entropy: Arc::new(engram_core::traits::OsEntropy), - }; - let cfg = CoordinatorConfig { - local_path: tmp, - ..CoordinatorConfig::default() - }; - ( - Arc::new(AppState::new_with_registry(cfg, services, host_registry)), - meta, - ) - } - - /// ADR 0045 C1 / ADR 0079: a RUNNING op on the session (a live - /// migration's inline claim, a peer pod's pipeline) makes the - /// scanner SKIP the session this tick — no transition, no restore - /// attempt, no double-driving. - #[tokio::test] - async fn advance_one_skips_when_an_op_owns_the_session() { - let session = evacuating_session(None); - let session_id = session.id; - let (state, meta) = build_state(session.clone()); - meta.ops - .seed_running(session_id, engram_core::types::session_op::OpKind::Teleport); - - advance_one(&EvacResumerConfig::default(), &state, session, 0) - .await - .expect("skip is not an error"); - - let after = meta.get_session(session_id).await.unwrap(); - assert_eq!( - after.status, - SessionState::Evacuating, - "an op-owned session must be left untouched for the holder", - ); - } - - /// No snapshot + no live disk manifest → NoRecoverableState → - /// Dead on attempt 1. - #[tokio::test] - async fn structural_no_state_fails_fast_to_dead() { - let session = evacuating_session(None); - let session_id = session.id; - let (state, meta) = build_state(session.clone()); - - advance_one(&EvacResumerConfig::default(), &state, session, 0) - .await - .expect("advance_one swallows structural failures"); - - let after = meta.get_session(session_id).await.unwrap(); - assert_eq!( - after.status, - SessionState::Dead, - "structurally unrecoverable session must fail fast to Dead, not retry", - ); - } - - /// Live disk manifest present but the image is not enabled (no - /// cold-boot spec derivable) → ColdBootUnavailable → Idle on - /// attempt 1, preserving the re-enable-then-/resume path. - #[tokio::test] - async fn structural_disk_only_without_image_fails_fast_to_idle() { - let live = engram_core::types::manifest::ManifestRef { - manifest_id: uuid::Uuid::from_u128(0xD15C), - version: 4, - }; - let session = evacuating_session(Some(live)); - let session_id = session.id; - let (state, meta) = build_state(session.clone()); - - advance_one(&EvacResumerConfig::default(), &state, session, 0) - .await - .expect("advance_one swallows structural failures"); - - let after = meta.get_session(session_id).await.unwrap(); - assert_eq!( - after.status, - SessionState::Idle, - "disk-only session with no enabled image must land at Idle \ - (re-enable image + /resume recovers), not burn the budget", - ); - } - - /// Issue #214: the aged-pin predicate. A fresh stamp is honored; one - /// past the TTL is a leak; a `None` stamp (legacy pin) is always - /// honored so the rollout doesn't drop in-flight pins. - #[test] - fn teleport_pin_aged_respects_ttl() { - let now = Utc::now(); - assert!( - !teleport_pin_aged(Some(now), now), - "a just-set pin is not aged", - ); - assert!( - !teleport_pin_aged( - Some(now - (TELEPORT_PIN_TTL - chrono::Duration::seconds(1))), - now - ), - "a pin just inside the TTL is honored", - ); - assert!( - teleport_pin_aged( - Some(now - (TELEPORT_PIN_TTL + chrono::Duration::seconds(1))), - now - ), - "a pin past the TTL is treated as a leak", - ); - assert!( - !teleport_pin_aged(None, now), - "a NULL stamp (pre-0065 legacy pin) is never aged out", - ); - } - - /// Issue #214 acceptance: a teleport pin older than the TTL is a leak. - /// The evac scanner must IGNORE it (not strictly pin this evacuation - /// onto the stale target) and CLEAR it (so it can't hijack a future - /// evacuation either). We observe the clear directly; the "ignore" is - /// implied because a structurally-recoverable session with a honored - /// pin to an unknown host would burn its budget, whereas this session - /// resolves on attempt 1 (structural fail-fast) with the pin gone. - #[tokio::test] - async fn aged_teleport_pin_is_ignored_and_cleared() { - let session = evacuating_session(None); - let session_id = session.id; - let (state, meta) = build_state(session.clone()); - - // Plant an aged pin straight into the mirror: a real leak from - // some path that set the pin >TTL ago without the session ever - // resolving off Evacuating. The set-at is backdated past the TTL. - let stale_target = engram_core::HostId::new(); - meta.teleport_targets.lock().insert( - session_id, - ( - stale_target, - Some(Utc::now() - (TELEPORT_PIN_TTL + chrono::Duration::minutes(1))), - ), - ); - - // No recoverable state → the pipeline fails fast to Dead, but the - // aged-pin check runs first (before evacuate_dead_source). - advance_one(&EvacResumerConfig::default(), &state, session, 0) - .await - .expect("advance_one swallows structural failures"); - - assert_eq!( - meta.get_teleport_target(session_id).await.unwrap(), - None, - "an aged (leaked) teleport pin must be cleared, not left to hijack \ - a future evacuation", - ); - } - - /// Inverse of the above: a FRESH pin must NOT be cleared by the - /// aged-pin guard — it is consumed normally by the resolution path. - /// (Here the session fails structurally, so the pin is cleared by the - /// terminal-fail arm rather than the aged-pin arm; the point is the - /// aged-pin arm doesn't fire and warn on a healthy pin.) We assert the - /// predicate path: a freshly-stamped pin is observed as require_host. - #[test] - fn fresh_teleport_pin_is_not_aged() { - assert!( - !teleport_pin_aged(Some(Utc::now()), Utc::now()), - "a freshly-set teleport pin must be honored, never aged out", - ); - } -} diff --git a/crates/engram-coordinator/src/evacuation.rs b/crates/engram-coordinator/src/evacuation.rs deleted file mode 100644 index a72b3e2a8..000000000 --- a/crates/engram-coordinator/src/evacuation.rs +++ /dev/null @@ -1,1421 +0,0 @@ -//! ADR 0018: dead-source session evacuation primitive. -//! -//! `evacuate_dead_source` restores a session onto a peer host from -//! already-recorded artifacts (a snapshot row and/or the live disk -//! manifest) when the source host is gone. Its sole caller is the -//! `evac_resumer` background scanner (driving `Evacuating → Created`), -//! fed by operator drain (ADR 0044 K3). ADR 0045 Phase A retired the -//! reactive NBD-loss / dead-host producers, so this is a drain-only -//! primitive now. -//! -//! The primitive leaves the session at `Created` on the new host: the -//! restored VM has snapshotted memory but a stale harness (its vsock -//! to the source's agentd died with the original sandbox). Callers -//! finish the resume-shape dance by running `start_agent` against the -//! new sandbox + transitioning to `Active`, exactly the way -//! `resume_from_fc_snapshot` in `api/snapshot.rs` finishes a resume. -//! Keeping that step outside the primitive is what lets it be unit- -//! tested against mocks without spinning up SharedState. -//! -//! ADR 0018 commit 12h retired the synchronous *alive-source* -//! `evacuate_to` primitive: the async rework (Evacuating state + -//! scanner) made it dead code. All evac now flows through the -//! state-machine + `evacuate_dead_source` resume path. - -use std::sync::Arc; - -use engram_core::traits::{MetadataStore, SessionFence}; -use engram_core::types::evacuation::{EvacLoss, EvacReceipt}; -use engram_core::types::manifest::ManifestRef; -use engram_core::types::session::{Session, SessionState}; -use engram_core::types::snapshot::{SnapshotMetadata, SnapshotRecord}; -use engram_core::types::BindingDisposition; -use engram_core::{MetaError, SandboxError}; - -use crate::host_registry::HostRegistry; -use crate::placement::{PickError, ScheduleContext}; - -/// Errors specific to the evacuation primitive. Wraps the upstream -/// `SandboxError` / `MetaError` so callers can distinguish which step -/// failed for retry / telemetry decisions. -#[derive(Debug)] -pub enum EvacError { - RestoreFailed(SandboxError), - Rebind(MetaError), - /// No recoverable state to restore from. Dead-source path returns - /// this when both `snapshot` and `session.live_disk_manifest` are - /// `None`. Caller routes to `HostLost → Dead`. - NoRecoverableState, - /// ADR 0028 Fix B: the session is disk-only recoverable (a live - /// disk manifest exists but no usable snapshot) and the cold-boot - /// spec resolved to `None` — the session's image is no longer - /// enabled, so there's no manifest to derive boot resources from. - /// Structural: retrying won't fix it; the caller routes to `Idle` - /// (re-enable the image, then `/resume` recovers via the same - /// cold-boot path). - ColdBootUnavailable(String), - /// ADR 0116: the lazily-awaited cold-boot materialization failed — - /// a transient store/catalog read inside `materialize_cold_boot` - /// (harness selection, runtime spec, fleet catalog, …). Retryable, - /// NOT structural: the next scanner tick / resume retry re-derives - /// it. The future is awaited only on the disk-only rung, so a - /// memory-snapshot recovery can never fail on slot resolution. - SpecResolution(crate::error::ApiError), - /// No host could accept the relocate (no capacity, or no host - /// with the image prefetched). Caller logs + retries later or - /// routes to `HostLost → Dead`. - NoTargetAvailable(PickError), - /// #800 (RESERVED evac placement): no survivor FITS the session's - /// reserved 2D budget. Distinct from `NoTargetAvailable` (a transient - /// pick failure the resumer retries against): this is the honest - /// hard-bound overflow — the caller QUEUES the session (`Evacuating → - /// Queued`, resume-origin) rather than binding a measured-full host, - /// and the queue scanner re-homes it once capacity returns. Not - /// structural (capacity does return), but also not a per-tick retry - /// (queueing hands ownership to the scanner). - NoCapacityQueue, -} - -impl EvacError { - /// ADR 0028 Fix B fail-fast guard: structural errors can never be - /// fixed by retrying — burning the resumer's 20-attempt budget on - /// them (the `cf4d4afd` incident's ~3 min of `RestoreFailed` - /// churn) just delays the honest terminal state. - pub fn is_structural(&self) -> bool { - matches!( - self, - Self::NoRecoverableState | Self::ColdBootUnavailable(_) - ) - } -} - -impl std::fmt::Display for EvacError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::RestoreFailed(e) => write!(f, "target-side restore failed: {e}"), - Self::Rebind(e) => write!(f, "PG rebind failed: {e}"), - Self::NoRecoverableState => { - write!( - f, - "no snapshot or live disk manifest — session cannot be evacuated" - ) - } - Self::ColdBootUnavailable(reason) => { - write!( - f, - "disk-only recoverable but no cold-boot spec available: {reason}" - ) - } - Self::SpecResolution(e) => { - write!(f, "cold-boot spec resolution failed (retryable): {e}") - } - Self::NoTargetAvailable(e) => write!(f, "no host could accept the relocate: {e:?}"), - Self::NoCapacityQueue => write!( - f, - "no survivor fits the session's reserved budget — queue instead of overcommit" - ), - } - } -} - -impl std::error::Error for EvacError { - fn source(&self) -> Option<&(dyn std::error::Error + 'static)> { - match self { - Self::NoRecoverableState - | Self::ColdBootUnavailable(_) - | Self::NoTargetAvailable(_) - | Self::NoCapacityQueue => None, - Self::RestoreFailed(e) => Some(e), - Self::Rebind(e) => Some(e), - Self::SpecResolution(e) => Some(e), - } - } -} - -// ADR 0028 Fix B's spec derivation moved to -// `crate::boot_materializer::materialize_cold_boot` (ADR 0116): the -// disk-only recovery spec now also carries the session's persisted slot -// selections (harness/skills), so a recovered guest can always spawn what -// its argv names. Callers pass the materialization as an UN-AWAITED -// future; `evacuate_dead_source` awaits it only on the disk-only rung -// (`SpecResolution` = transient read, retryable; `ColdBootUnavailable` = -// image un-enabled, structural). - -/// Pick the disk manifest the target should restore from. Mirrors the -/// `effective_resume_disk_manifest` semantics in `api/snapshot.rs`: -/// when both live and snapshot manifests exist, prefer the live one -/// only if it's a strictly newer version of the same manifest_id -/// lineage; otherwise the snapshot wins (different lineage means we -/// trust the (memory, disk) pair the snapshot captured together). -fn pick_evac_disk_manifest( - live: Option, - snapshot: Option, -) -> Option { - match (live, snapshot) { - (None, snap) => snap, - (Some(l), None) => Some(l), - (Some(l), Some(s)) => { - if l.manifest_id == s.manifest_id && l.version > s.version { - Some(l) - } else { - Some(s) - } - } - } -} - -/// Mechanics for **dead-source** evacuation. Driven by the -/// `evac_resumer` scanner for `Evacuating` sessions (produced by -/// operator drain, ADR 0044 K3) — restores from existing artifacts -/// because the source backend is unreachable, so a fresh source-side -/// snapshot isn't possible. (ADR 0045 Phase A retired the reactive -/// dead-host / NBD-loss producers that used to feed this.) -/// -/// Restores from existing artifacts, rung-aware (ADR 0028 recovery -/// ladder): -/// -/// - **Rung 1 — coherent checkpoint** (`snapshot.memory_manifest` -/// present): restore memory + the checkpoint's OWN -/// `snapshot.disk_manifest`. Deliberately NOT the newer -/// `live_disk_manifest`: the restored RAM describes the checkpoint's -/// disk, so pairing it with a later disk version is incoherent. This -/// IS the "rewind the disk to the checkpoint" step — post-checkpoint -/// continuous-sync deltas are discarded for coherence. `EvacLoss::None`. -/// - **Rung 2 — cold boot** (no memory): a fresh kernel mounts the -/// `pick_evac_disk_manifest(live, snapshot)` disk (live-wins-when- -/// newer — newest files, no coherence constraint since there's no -/// RAM to disagree). `EvacLoss::Memory { reason: "source-dead-no-snapshot" }`. -/// -/// Failure modes: -/// -/// - `snapshot.is_none() && session.live_disk_manifest.is_none()` → -/// `NoRecoverableState`. Caller drives `HostLost → Dead`. -/// - Target pick fails (no capacity, image not ready) → -/// `NoTargetAvailable`. Caller logs + may retry; nothing changed. -/// - Target-side restore fails → `RestoreFailed`. Caller may retry -/// against a different host; nothing changed. -/// - PG rebind fails after restore → new sandbox is up on target -/// without a PG row pointing at it (a small orphan window). Idle -/// evictor / host-agent restart sweep reaps. -/// -/// Leaves the session at `Created` on the new host. Caller is -/// responsible for the start_agent + Active transition. -#[allow(clippy::too_many_arguments)] // cohesive relocation inputs; threading a struct buys nothing -pub async fn evacuate_dead_source( - registry: &Arc, - meta: &Arc, - session: Session, - snapshot: Option, - // ADR 0028 Fix B / ADR 0116: the disk-only recovery's boot shape, - // passed as an UN-AWAITED future (`boot_materializer:: - // materialize_cold_boot`) and awaited only on the disk-only rung. - // The rung decision lives HERE, with the recovery ladder — so a - // caller can never eagerly fail a memory-snapshot recovery on slot - // resolution it would not consult, and never has to pick an error - // posture for a value it doesn't know is needed. `Ok(None)` (image - // un-enabled) fails structurally with `ColdBootUnavailable`; `Err` - // (transient store read) fails retryably with `SpecResolution`. - cold_boot_spec: F, - // ADR 0045 Phase F (teleport): when `Some`, place onto this exact - // host instead of the capacity-ranked pick (operator-pinned - // destination). `None` keeps the standard any-peer policy. - require_host: Option, - // ADR 0045 C2 (E2B fold, origin affinity): soft preference for this - // host in the capacity-ranked pick — tier-2, below snapshot - // affinity, loses to exclude/draining/capacity (host_registry's - // standard precedence). The disk-only RESUME path passes the - // session's ORIGIN host (its chunk cache + base shm are warm - // there); movers (drain, dead-host) pass `None` — they are moving - // AWAY by definition. - prefer_host: Option, - // ADR 0079: the caller's op/claim epoch, stamped into the restore - // RPC (the evac-resumer claim, the resume verb's disk-only path). - fence: SessionFence, - // #800 (RESERVED evac placement): the session's reserved 2D budget - // `(mem_mib, cpu_vcpus)`, resolved from the enabled image - // (`boot_materializer::materialize_cold_boot`). `Some` feeds the HARD reserved pick — a - // relocation that fits no survivor returns `NoCapacityQueue` (the caller - // queues instead of overcommitting). `None` (image un-enabled / budget - // unresolvable) keeps the pre-#800 capacity-SOFT posture — never strand - // an evacuation on a spec-resolution blip (mirrors the resume verb's - // budget-unresolved fallback in `resume_from_fc_snapshot`). - budget: Option<(u32, u32)>, - // ADR 0098 D1: the caller's injected wall clock, used for the - // peer-hint heartbeat-staleness gate (`host_can_serve_chunks`). - now: chrono::DateTime, -) -> Result -where - F: std::future::Future< - Output = Result, crate::error::ApiError>, - >, -{ - let session_id = session.id; - let old_sandbox_id = session.sandbox_id; - - let memory_manifest = snapshot.as_ref().and_then(|s| s.memory_manifest); - let snapshot_disk = snapshot.as_ref().and_then(|s| s.disk_manifest); - - // ADR 0028 rung-aware disk pick — the coherence rule. With a - // coherent memory snapshot (rung 1), the restored RAM's page - // cache + mounted-fs metadata describe the checkpoint's OWN disk; - // pairing it with the newer continuous-sync `live_disk_manifest` - // is incoherent (corruption — Defect B in miniature). So rung 1 - // uses the snapshot's disk verbatim. Only rung 2 (cold boot, no - // memory — a fresh kernel mounts whatever it's given) takes the - // live-wins preference. - let disk_manifest = if memory_manifest.is_some() { - snapshot_disk - } else { - pick_evac_disk_manifest(session.live_disk_manifest, snapshot_disk) - }; - - if disk_manifest.is_none() && memory_manifest.is_none() { - return Err(EvacError::NoRecoverableState); - } - - let (loss, reason) = if memory_manifest.is_some() { - (EvacLoss::None, "") - } else { - ( - EvacLoss::Memory { - reason: "source-dead-no-snapshot".into(), - }, - "source-dead-no-snapshot", - ) - }; - let _ = reason; // structured-log placeholder; metric label lives on `loss.as_str()`. - - // ADR 0028 Fix B: validate the disk-only branch's prerequisites - // BEFORE the host pick. Structural failures (no cold-boot spec) - // must not hide behind transient ones (`NoTargetAvailable`) — - // otherwise a capacity blip masks an unrecoverable session and - // the resumer retries something retrying can't fix. - let cold_boot = if memory_manifest.is_some() { - None - } else { - let disk = disk_manifest - .expect("disk-only branch requires a disk manifest (NoRecoverableState guards above)"); - // ADR 0116: first (and only) await of the lazily-passed - // materialization — the memory-snapshot rung above never runs it. - let mut spec = cold_boot_spec - .await - .map_err(EvacError::SpecResolution)? - .ok_or_else(|| { - EvacError::ColdBootUnavailable(format!( - "session {session_id} has a live disk manifest ({disk}) but no \ - cold-boot spec — is its image still enabled?" - )) - })?; - spec.rootfs_manifest = Some(disk); - Some(spec) - }; - - let (image_repo, image_tag) = engram_core::types::session::split_image_ref(&session.image); - let ctx = ScheduleContext { - repo: image_repo, - image_version: image_tag, - snapshot_host: snapshot.as_ref().and_then(|s| s.host_id), - // #800: feed the reserved budget (was hard-coded `None`, - // capacity-blind — the ADR 0046 evac-leg gap). With it set, the - // tier-0 snapshot-host / tier-2 prefer vetoes AND best-fit all gate - // on the real 2D budget, and the HARD reserved pick can honestly - // report "nothing fits" instead of soft-binding a full survivor. - memory_mib: budget.map(|(mib, _)| mib), - cpu_budget_vcpus: budget.map(|(_, vcpus)| vcpus), - // Target-selection: image-cache-warm preference is a future - // refinement (defer when we add zone tagging to HostState). - // Today we accept any host that can take the work, but never - // the source host (exclude_host) — set to the prior owner via - // `session.host_id` so a drain doesn't relocate back onto the - // host being drained. For the dead-source path the source is - // already unregistered; exclude_host is defensive. - required_image_digest: None, - exclude_host: session.host_id, - prefer_host, - // ADR 0068: same pairing the resume path uses — a memory - // manifest needs the FC UFFD substrate, and (when known) the - // target must match the snapshot's capture-time - // `fc_snapshot_version` exactly. - caps: crate::placement::CapabilityRequirements { - needs_uffd_substrate: memory_manifest.is_some(), - fc_snapshot_version: snapshot - .as_ref() - .and_then(|s| s.fc_snapshot_version.clone()), - }, - // ADR 0090: steer the relocation toward hosts whose bundle stamp - // already covers the snapshot's pinned generations (campaign B1: - // a recovery landed on a mid-staging fresh node and the harness - // spawn had nothing to exec). Soft — see rank_hosts. - prefer_bundles: snapshot - .as_ref() - .map(|s| s.aux_bundles.as_slice()) - .unwrap_or(&[]), - }; - - // Split pick + restore so picker errors and backend errors keep - // distinct typing — picker failures are `NoTargetAvailable` - // (operator action: free capacity or wait for image prefetch); - // backend failures are `RestoreFailed` (retry against another - // host or surface to user). ADR 0045 Phase F: an operator-pinned - // teleport target bypasses capacity ranking and places on that - // exact host (still excluding the source); a bad pin retries then - // falls back to Idle rather than silently landing elsewhere. - let (target_host, target_backend) = match require_host { - Some(host) => crate::placement::pick_specific_host( - meta.as_ref(), - registry, - host, - session.host_id, - now, - ) - .await - .map_err(EvacError::NoTargetAvailable)?, - // #800 (RESERVED evac placement): the standard any-peer path honors - // the HARD reserved 2D bound when a budget is known — a relocation - // that fits no survivor returns `NoCapacity`, which we surface as - // `NoCapacityQueue` so the resumer QUEUES the session (resume-origin) - // instead of binding a measured-full host (the #722/#795 - // over-reservation class, on the evac leg — issue #800). Other pick - // errors stay `NoTargetAvailable` (transient — the resumer retries). - // A `None` budget (image un-enabled / unresolvable) keeps the - // pre-#800 capacity-soft pick — never strand on a spec-resolution - // blip. (An operator teleport pin — the `Some(host)` arm above — - // stays capacity-soft by design; an explicit pin overrides ranking.) - None if budget.is_some() => { - match crate::placement::pick_for_session_reserved(meta.as_ref(), registry, &ctx, now) - .await - { - Ok(picked) => picked, - Err(PickError::NoCapacity) => return Err(EvacError::NoCapacityQueue), - Err(e) => return Err(EvacError::NoTargetAvailable(e)), - } - } - None => crate::placement::pick_for_session(meta.as_ref(), registry, &ctx, now) - .await - .map_err(EvacError::NoTargetAvailable)?, - }; - - let new_sandbox_id = match cold_boot { - // ADR 0028 Fix B — rung 2: no coherent memory snapshot, but - // the continuous-sync disk manifest is current. A full-FC - // restore is structurally impossible here (no state.bin, no - // sidecar — the pre-Fix-B code minted a nil-blob-key - // SnapshotId and burned the resumer's whole budget on - // "manifest.json: No such file or directory"). And pairing - // the image's BASE memory with this *evolved* disk would be - // incoherent (restored RAM's page cache + mounted-fs metadata - // describe the base disk → corruption). The coherent recovery - // is a FRESH KERNEL BOOT mounting the recovered rootfs: the - // live manifest is a full rootfs lineage, so the host - // NBD-attaches it and boots clean. On-disk work survives; - // in-RAM context does not (`EvacLoss::Memory`). - Some(spec) => target_backend - .create(spec) - .await - .map_err(EvacError::RestoreFailed)?, - // Rung-1-shaped recovery: a coherent memory snapshot exists. - // Lift the snapshot row's fields verbatim (id, size, - // image_version) and DERIVE the portable blob keys from the - // snapshot_id. The keys are deterministic functions of the id - // (`snapshots//{state.bin,sidecar.json}`), so we can - // rebuild them at restore time without needing dedicated - // columns on the snapshot row. - // - // ADR 0018 commit 12 (async evac): the source records the - // snapshot in PG before the scanner picks the session up, - // which loses the `state_blob_key` / `sidecar_blob_key` that - // `PooledBackend::snapshot` populated on the in-memory - // metadata. Deriving them here is the canonical fix — same - // shape as the matching `materialize_state_if_missing` helper - // that consumes them on the target. Without these, the target - // host's FC `restore()` errors with "manifest.json: No such - // file or directory" because nothing materialised the sidecar. - None => { - let s = snapshot - .as_ref() - .expect("memory_manifest implies a snapshot row"); - // ADR 0045 D4: best-effort image-base canonical ref (shared - // per-image base shm on the target); None falls back to - // canonical == session. - let base_memory_manifest = match meta.get_enabled_image(&session.image).await { - Ok(Some(img)) => match img.base_snapshot_id { - Some(bid) => meta - .get_snapshot(bid) - .await - .ok() - .flatten() - .and_then(|b| b.memory_manifest), - None => None, - }, - _ => None, - }; - // ADR 0095: the snapshot-rehome leg of an evacuation is - // exactly the peer-fill case — the capturing host is - // draining (cordoned) but ALIVE, one LAN hop away, with the - // whole divergent set on its NVMe. Hint it - // (`host_can_serve_chunks` deliberately ignores the - // cordon); a dead source degrades to the GCS path. - let peer_hints = match s.host_id { - Some(src) => match meta.list_active_hosts().await { - Ok(hosts) => { - let ttl = crate::placement::placement_ttl(); - hosts - .iter() - .find(|h| h.id == src) - .and_then(|h| crate::placement::host_can_serve_chunks(h, now, ttl)) - .map(|addr| vec![addr.to_string()]) - .unwrap_or_default() - } - Err(_) => Vec::new(), - }, - None => Vec::new(), - }; - let metadata = SnapshotMetadata { - migration_source: None, - id: s.id, - size_bytes: s.size_bytes, - created_at: s.created_at, - image_version: s.image_version.clone(), - disk_manifest, - memory_manifest, - base_memory_manifest, - source_sandbox_id: None, - state_blob_key: Some(engram_chunk_store::snapshot_blob::state_blob_key(s.id)), - sidecar_blob_key: Some(engram_chunk_store::snapshot_blob::sidecar_blob_key(s.id)), - rootfs_blob_key: None, - working_set_blob_key: None, - // ADR 0035: evac-dest restore is resume-flavored — keep the - // pinned generations; the target host materializes them. - aux_bundles: s.aux_bundles.clone(), - // Issue #529: restore-side reconstruction, not a fresh capture. - paused_at: None, - // ADR 0095: assembled above — the draining source, or empty. - peer_hints, - }; - target_backend - .restore(metadata, fence) - .await - .map_err(EvacError::RestoreFailed)? - } - }; - - // Routing cache: invalidate the stale source binding (the source - // host is dead, so this is usually already gone from - // `host_registry.unregister`, but defensive). Insert the new. - if let Some(old) = old_sandbox_id { - registry.invalidate_sandbox(old); - } - registry.record_sandbox_owner(new_sandbox_id, target_host); - - // PG rebind. The `evac_resumer` scanner has the session at - // `Evacuating` (operator drain flipped Active → Evacuating), so we - // drive Evacuating → Created here. (`transition_session` enforces - // the legality table; both Evacuating → Created and the legacy - // HostLost → Created are legal edges.) - meta.assign_session_host(session_id, Some(target_host)) - .await - .map_err(EvacError::Rebind)?; - let binding_epoch = meta - .assign_session_sandbox(session_id, Some(new_sandbox_id)) - .await - .map_err(EvacError::Rebind)? - .ok_or_else(|| { - EvacError::Rebind(engram_core::MetaError::Serialization( - "binding write returned no epoch".into(), - )) - })?; - meta.transition_session( - session_id, - SessionState::Created, - BindingDisposition::Retain, - ) - .await - .map_err(EvacError::Rebind)?; - - Ok(EvacReceipt { - binding_epoch, - new_host_id: target_host, - new_sandbox_id, - loss, - }) -} - -#[cfg(test)] -// tests drive a live system; wall clock/OS entropy here is input, not a decision source (ADR 0098 D1) -#[allow(clippy::disallowed_methods)] -mod tests { - use super::*; - use async_trait::async_trait; - use engram_core::traits::{HarnessDial, HostClient}; - use engram_core::types::sandbox::{ExecRequest, ExecStream, SandboxSpec}; - use engram_core::types::session::{Session, SessionMode, SessionState}; - use engram_core::types::snapshot::SnapshotMetadata; - use engram_core::{HostId, SandboxId, SessionId}; - use parking_lot::Mutex as PlMutex; - use std::sync::atomic::{AtomicUsize, Ordering}; - - /// Records every call made against it; deterministic returns so - /// the orchestration's call ordering is observable in tests. - /// Snapshot returns a SnapshotMetadata with deterministic UUID; - /// restore returns the pre-staged `next_restore_id`. - #[derive(Default)] - struct FakeBackend { - next_restore_id: PlMutex>, - fail_snapshot: AtomicUsize, // 0=ok, 1=fail - fail_restore: AtomicUsize, - /// ADR 0028 Fix B: the spec the disk-only cold-boot branch - /// passed to `create()`, for assertions. - last_create_spec: PlMutex>, - /// ADR 0028 rung-1: the metadata the warm branch passed to - /// `restore()`, for the coherence-rule assertions. - last_restore_metadata: PlMutex>, - } - - impl FakeBackend { - fn set_restore_id(&self, id: SandboxId) { - *self.next_restore_id.lock() = Some(id); - } - - fn last_restore_metadata(&self) -> Option { - self.last_restore_metadata.lock().clone() - } - - fn last_create_spec(&self) -> Option { - self.last_create_spec.lock().clone() - } - } - - #[async_trait] - impl HostClient for FakeBackend { - async fn create(&self, spec: SandboxSpec) -> Result { - *self.last_create_spec.lock() = Some(spec); - // Reuse the restore-id knob so disk-only tests can pin - // the expected sandbox id; fresh id otherwise (match-based - // to dodge the unwrap_or_default lint — a nil-UUID default - // would mask test bugs). - Ok(match *self.next_restore_id.lock() { - Some(id) => id, - None => SandboxId::new(), - }) - } - async fn destroy(&self, _id: SandboxId, _fence: SessionFence) -> Result<(), SandboxError> { - Ok(()) - } - async fn list(&self) -> Result, SandboxError> { - Ok(vec![]) - } - async fn probe_sandbox( - &self, - _id: SandboxId, - ) -> Result { - unimplemented!() - } - async fn exec_stream( - &self, - _id: SandboxId, - _cmd: ExecRequest, - ) -> Result { - unreachable!() - } - async fn snapshot( - &self, - _id: SandboxId, - _fence: SessionFence, - ) -> Result { - if self.fail_snapshot.load(Ordering::SeqCst) > 0 { - return Err(SandboxError::Vm(Box::new(SimpleErr( - "snapshot failed".into(), - )))); - } - Ok(SnapshotMetadata { - id: engram_core::SnapshotId::new(), - size_bytes: 1024, - created_at: chrono::Utc::now(), - image_version: "test".into(), - base_memory_manifest: None, - migration_source: None, - disk_manifest: None, - memory_manifest: None, - source_sandbox_id: None, - state_blob_key: None, - sidecar_blob_key: None, - rootfs_blob_key: None, - working_set_blob_key: None, - aux_bundles: vec![], - paused_at: None, - peer_hints: Vec::new(), - }) - } - async fn commit_snapshot( - &self, - _id: SandboxId, - _fence: SessionFence, - ) -> Result<(), SandboxError> { - Ok(()) - } - async fn abort_snapshot( - &self, - _id: SandboxId, - _fence: SessionFence, - ) -> Result<(), SandboxError> { - Ok(()) - } - async fn restore( - &self, - md: SnapshotMetadata, - _fence: SessionFence, - ) -> Result { - *self.last_restore_metadata.lock() = Some(md); - if self.fail_restore.load(Ordering::SeqCst) > 0 { - return Err(SandboxError::Vm(Box::new(SimpleErr( - "restore failed".into(), - )))); - } - // FakeBackend tests always preload a restore id; the - // fallback is just defensive against a misconfigured test. - // Lifted out of unwrap_or_else / unwrap_or to dodge clippy's - // unwrap_or_default lint (Default would mint a nil UUID, - // which would mask test bugs vs. a fresh id flagging them). - Ok(match *self.next_restore_id.lock() { - Some(id) => id, - None => SandboxId::new(), - }) - } - async fn start_agent( - &self, - _id: SandboxId, - _agent: engram_core::types::sandbox::AgentSpec, - _policy: engram_core::types::egress::SessionEgressPolicy, - _fence: SessionFence, - ) -> Result<(), SandboxError> { - unreachable!("evac primitive does not call start_agent") - } - async fn guest_ip(&self, _id: SandboxId) -> Option { - None - } - async fn bind_session( - &self, - _session_id: SessionId, - _sandbox_id: SandboxId, - _binding_epoch: u64, - ) -> Result<(), engram_core::SandboxError> { - Ok(()) - } - async fn unbind_session(&self, _session_id: SessionId) {} - async fn send_prompt( - &self, - _sandbox_id: SandboxId, - _prompt_id: String, - _text: String, - _mode: Option, - ) -> Result<(), SandboxError> { - unreachable!() - } - fn harness_dial(&self) -> HarnessDial { - HarnessDial::Vsock - } - } - - #[derive(Debug)] - struct SimpleErr(String); - impl std::fmt::Display for SimpleErr { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - f.write_str(&self.0) - } - } - impl std::error::Error for SimpleErr {} - - /// Minimal MetadataStore that records every state-affecting call. - /// Stage a HostLost session through REAL store calls (ADR 0098: - /// FakeMeta retired onto SimMetadataStore) — create → bind → - /// Active → optional live-manifest publish → HostLost. Every hop - /// is a legal FSM edge, so the fixture can't stage a state the - /// production write paths couldn't reach. - async fn stage_hostlost_session( - meta: &Arc, - host: HostId, - sandbox: SandboxId, - live_disk: Option, - ) -> Session { - let id = meta - .create_session(engram_core::types::session::SessionSpec { - image: "ghcr.io/test/img:t".into(), - mode: SessionMode::Agent, - }) - .await - .expect("create"); - meta.assign_session_host(id, Some(host)) - .await - .expect("host"); - meta.transition_session_created(id, sandbox) - .await - .expect("created"); - meta.transition_session(id, SessionState::Active, BindingDisposition::Retain) - .await - .expect("active"); - if let Some(m) = live_disk { - let out = meta - .update_live_disk_manifest(id, sandbox, m) - .await - .expect("live manifest"); - assert!(matches!(out, engram_core::traits::UpdateOutcome::Applied)); - } - meta.transition_session(id, SessionState::HostLost, BindingDisposition::Retain) - .await - .expect("host_lost"); - meta.get_session(id).await.expect("staged") - } - - fn sim_meta() -> Arc { - engram_sim::SimMetadataStore::new( - Arc::new(engram_core::traits::SystemClock::new()), - Arc::new(engram_sim::SimEntropy::seeded(0xE7AC)), - ) - } - - /// Stage a fresh-heartbeat `ready` host row — deliberately - /// UNMEASURED (no utilization heartbeat), matching the retired - /// FakeMeta fixture: placement admits it via the last-resort - /// unmeasured tier, which is exactly the capacity-fallback path - /// these evac tests exercise. - async fn add_ready_host(meta: &Arc, id: HostId) { - let now = engram_core::traits::Clock::now_utc(&engram_core::traits::SystemClock::new()); - meta.upsert_host(engram_core::types::host::HostRecord { - id, - hostname: format!("fake-{id}"), - cloud_metadata: Default::default(), - capacity: engram_core::types::host::HostCapacity { - total_gb: 0, - used_gb: 0, - total_mib: 16_384, - used_mib: 0, - running_sandboxes: 0, - }, - utilization: Default::default(), - status: engram_core::types::host::HostStatus::Ready, - last_heartbeat_at: now, - host_addr: None, - ready_images: Vec::new(), - current_bundles: Vec::new(), - sandbox_bundles: Vec::new(), - cordoned: false, - cordon_owner: None, - cordon_reason: None, - retire_requested_at: None, - retired_at: None, - total_vcpus: 0, - wire_version: 0, - stages_images: false, - capabilities: engram_core::types::host::HostCapabilities::default(), - lease_expires_at: None, - lease_state: Default::default(), - lease_epoch: 0, - }) - .await - .expect("host row"); - } - - /// #800: a MEASURED host — `allocatable_mib` set — so placement's 2D - /// fit actually gates (unlike `add_ready_host`'s unmeasured last-resort - /// tier). Used to exercise the RESERVED evac gate: an evac whose budget - /// exceeds `allocatable_mib` fits NO survivor and must queue. - async fn add_measured_host( - meta: &Arc, - id: HostId, - alloc_mib: u64, - ) { - let now = engram_core::traits::Clock::now_utc(&engram_core::traits::SystemClock::new()); - meta.upsert_host(engram_core::types::host::HostRecord { - id, - hostname: format!("measured-{id}"), - cloud_metadata: Default::default(), - capacity: engram_core::types::host::HostCapacity { - total_gb: 0, - used_gb: 0, - total_mib: 16_384, - used_mib: 0, - running_sandboxes: 0, - }, - utilization: engram_core::types::host::HostUtilization { - allocatable_mib: alloc_mib, - ..Default::default() - }, - status: engram_core::types::host::HostStatus::Ready, - last_heartbeat_at: now, - host_addr: None, - ready_images: Vec::new(), - current_bundles: Vec::new(), - sandbox_bundles: Vec::new(), - cordoned: false, - cordon_owner: None, - cordon_reason: None, - retire_requested_at: None, - retired_at: None, - total_vcpus: 4, - wire_version: 0, - stages_images: false, - capabilities: engram_core::types::host::HostCapabilities::default(), - lease_expires_at: None, - lease_state: Default::default(), - lease_epoch: 0, - }) - .await - .expect("measured host row"); - // The sim carries `allocatable_mib` via the HEARTBEAT, not - // `upsert_host` (which defaults utilization for a fresh row) — so - // stamp a heartbeat to make the host genuinely MEASURED. - meta.touch_host_heartbeat( - id, - engram_core::types::host::HostHeartbeat { - status: engram_core::types::host::HostStatus::Ready, - capacity: engram_core::types::host::HostCapacity { - total_gb: 0, - used_gb: 0, - total_mib: 16_384, - used_mib: 0, - running_sandboxes: 0, - }, - utilization: engram_core::types::host::HostUtilization { - allocatable_mib: alloc_mib, - ..Default::default() - }, - ready_images: Vec::new(), - current_bundles: Vec::new(), - sandbox_bundles: Vec::new(), - total_vcpus: 4, - wire_version: 0, - stages_images: false, - capabilities: engram_core::types::host::HostCapabilities::default(), - lease_renew_until: None, - }, - ) - .await - .expect("measured heartbeat"); - } - - fn make_snapshot_for( - session_id: SessionId, - disk: Option, - memory: Option, - ) -> SnapshotRecord { - SnapshotRecord { - id: engram_core::SnapshotId::new(), - session_id: Some(session_id), - host_id: None, - image_version: "test".into(), - size_bytes: 1024, - created_at: chrono::Utc::now(), - last_accessed_at: chrono::Utc::now(), - disk_manifest: disk, - memory_manifest: memory, - recoverable: true, - aux_bundles: vec![], - events_cursor: None, - fc_snapshot_version: None, - } - } - - fn fake_manifest(id: u128, version: u64) -> ManifestRef { - ManifestRef { - manifest_id: uuid::Uuid::from_u128(id), - version, - } - } - - /// Helper: build a HostRegistry + register one target host with - /// fresh capacity. Returns (registry, target_host, target_backend). - async fn build_registry_with_target( - meta: Arc, - ) -> (Arc, HostId, Arc) { - let registry = Arc::new(HostRegistry::new(meta.clone())); - let target_host = HostId::new(); - let target_be = Arc::new(FakeBackend::default()); - registry.register(target_host, target_be.clone()); - add_ready_host(&meta, target_host).await; - // pick_for_session's capacity fallback path requires a fresh - // host with no draining flag — register() sets defaults that - // suffice. - (registry, target_host, target_be) - } - - /// ADR 0028 rung-1 COHERENCE RULE: a coherent memory snapshot is - /// present AND the live disk manifest is strictly newer (same - /// lineage). The restore MUST use the checkpoint's OWN disk — NOT - /// the newer live one — because the restored RAM describes the - /// checkpoint's disk; pairing it with later disk deltas corrupts. - /// This is the "rewind the disk to the checkpoint" step. - /// - /// (Pre-Fix-A this arm used live-wins and was named - /// `..._uses_live_disk_when_newer` — that encoded the bug.) - #[tokio::test] - async fn evac_dead_source_rung1_uses_checkpoint_disk_not_newer_live() { - let meta = sim_meta(); - let lineage = 0xABCD; - let session = stage_hostlost_session( - &meta, - HostId::new(), - SandboxId::new(), - // Live disk is a strictly-newer version of the SAME lineage — - // exactly the case pick_evac_disk_manifest would prefer. - Some(fake_manifest(lineage, 5)), - ) - .await; - let session_id = session.id; - - let checkpoint_disk = fake_manifest(lineage, 3); - let memory = fake_manifest(lineage + 1, 1); - let snapshot = make_snapshot_for(session_id, Some(checkpoint_disk), Some(memory)); - - let (registry, target_host, target_be) = build_registry_with_target(meta.clone()).await; - let new_sandbox = SandboxId::new(); - target_be.set_restore_id(new_sandbox); - - let receipt = evacuate_dead_source( - ®istry, - &(meta.clone() as Arc), - session.clone(), - Some(snapshot), - poison_spec(), - None, - None, - SessionFence::unfenced(), - None, // #800: budget — tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await - .expect("rung-1 happy path"); - assert_eq!(receipt.new_host_id, target_host); - assert_eq!(receipt.new_sandbox_id, new_sandbox); - assert_eq!(receipt.loss, EvacLoss::None); - - // The coherence assertion: warm restore, checkpoint's own disk, - // checkpoint's memory — the newer live disk (v5) is ignored. - let md = target_be - .last_restore_metadata() - .expect("rung-1 must go through restore(), not create()"); - assert_eq!( - md.disk_manifest, - Some(checkpoint_disk), - "rung-1 must pair memory with the checkpoint's OWN disk (v3), \ - not the newer live disk (v5)", - ); - assert_eq!(md.memory_manifest, Some(memory)); - assert!( - target_be.last_create_spec().is_none(), - "rung-1 is a restore, never a cold-boot create", - ); - - let updated = meta.get_session(session_id).await.expect("session"); - assert_eq!(updated.status, SessionState::Created); - assert_eq!(updated.host_id, Some(target_host)); - assert_eq!(updated.sandbox_id, Some(new_sandbox)); - } - - /// Arm 2: snapshot present, no live_disk → restore uses the - /// snapshot's disk + memory. Loss=None. - #[tokio::test] - async fn evac_dead_source_uses_snapshot_when_no_live_disk() { - let meta = sim_meta(); - let session = stage_hostlost_session(&meta, HostId::new(), SandboxId::new(), None).await; - let session_id = session.id; - - let snapshot = make_snapshot_for( - session_id, - Some(fake_manifest(0x1234, 7)), - Some(fake_manifest(0x5678, 7)), - ); - - let (registry, _target_host, target_be) = build_registry_with_target(meta.clone()).await; - target_be.set_restore_id(SandboxId::new()); - - let receipt = evacuate_dead_source( - ®istry, - &(meta.clone() as Arc), - session.clone(), - Some(snapshot), - poison_spec(), - None, - None, - SessionFence::unfenced(), - None, // #800: budget — tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await - .expect("snapshot-only happy path"); - assert_eq!(receipt.loss, EvacLoss::None); - } - - /// ADR 0045 C2 (E2B fold, origin affinity): the disk-only RESUME - /// path threads the session's origin host as a soft preference — - /// its NBD chunk cache + base shm are warm there. Two otherwise - /// equal hosts: the pick lands on the preferred one. (Precedence — - /// snapshot affinity above, exclude/draining/capacity vetoes — is - /// pinned by host_registry's own prefer_host tests.) - #[tokio::test] - async fn evac_prefers_the_origin_host_when_passed() { - let meta = sim_meta(); - let staged = stage_hostlost_session(&meta, HostId::new(), SandboxId::new(), None).await; - // Mirrors resume_disk_only_cold_boot: host_id cleared (so the - // origin is NOT excluded), origin threaded as prefer_host. - meta.assign_session_host(staged.id, None) - .await - .expect("clear host"); - let session = meta.get_session(staged.id).await.expect("staged"); - let session_id = session.id; - let snapshot = make_snapshot_for( - session_id, - Some(fake_manifest(0x1111, 1)), - Some(fake_manifest(0x2222, 1)), - ); - - let registry = Arc::new(HostRegistry::new(meta.clone())); - let origin = HostId::new(); - let other = HostId::new(); - let origin_be = Arc::new(FakeBackend::default()); - let other_be = Arc::new(FakeBackend::default()); - registry.register(other, other_be.clone()); - registry.register(origin, origin_be.clone()); - add_ready_host(&meta, other).await; - add_ready_host(&meta, origin).await; - origin_be.set_restore_id(SandboxId::new()); - other_be.set_restore_id(SandboxId::new()); - - let receipt = evacuate_dead_source( - ®istry, - &(meta.clone() as Arc), - session.clone(), - Some(snapshot), - poison_spec(), - None, - Some(origin), - SessionFence::unfenced(), - None, // #800: budget — tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await - .expect("origin-affinity happy path"); - assert_eq!( - receipt.new_host_id, origin, - "with equal capacity the origin preference must win the pick" - ); - } - - /// A plausible cold-boot spec, the shape `materialize_cold_boot` - /// would derive from an enabled image. - /// ADR 0116: the lazily-passed cold-boot materialization, pre-resolved - /// for tests (production passes `materialize_cold_boot` un-awaited). - fn ready_spec( - spec: Option, - ) -> impl std::future::Future, crate::error::ApiError>> - { - std::future::ready(Ok(spec)) - } - - /// ADR 0116 (#1212 review finding): the memory-snapshot rung must not - /// RUN the cold-boot materialization — not merely tolerate its errors. - /// This future panics if anything ever polls it. - fn poison_spec( - ) -> impl std::future::Future, crate::error::ApiError>> - { - std::future::poll_fn(|_| { - panic!("cold-boot materialization was polled on the memory-snapshot rung") - }) - } - - fn test_cold_boot_spec() -> SandboxSpec { - SandboxSpec { - image: "ghcr.io/test/img:t".into(), - rootfs_source: None, - image_uri: Some("ghcr.io/test/img:t".into()), - rootfs_manifest: None, - cpu: engram_core::types::sandbox::CpuLimit { vcpus: 2 }, - memory: engram_core::types::sandbox::MemoryLimit { max_mib: 4096 }, - disk: engram_core::types::sandbox::DiskLimit { max_gib: 20 }, - ttl: None, - env: Default::default(), - workdir: None, - network: Default::default(), - aux_ro_drives: Vec::new(), - swap_mib: None, - } - } - - /// Arm 3 (ADR 0028 Fix B): no snapshot, live_disk present → - /// disk-only COLD BOOT — `create()` with the session's live - /// manifest as the rootfs override, NOT a structurally-impossible - /// `restore()`. Loss=Memory{reason}. - #[tokio::test] - async fn evac_dead_source_disk_only_cold_boots_with_memory_loss() { - let meta = sim_meta(); - let live = fake_manifest(0xCAFE, 9); - let session = - stage_hostlost_session(&meta, HostId::new(), SandboxId::new(), Some(live)).await; - let session_id = session.id; - - let (registry, _target_host, target_be) = build_registry_with_target(meta.clone()).await; - let new_sandbox = SandboxId::new(); - target_be.set_restore_id(new_sandbox); - - let receipt = evacuate_dead_source( - ®istry, - &(meta.clone() as Arc), - session.clone(), - None, - ready_spec(Some(test_cold_boot_spec())), - None, - None, - SessionFence::unfenced(), - None, // #800: budget — tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await - .expect("disk-only cold-boot happy path"); - match &receipt.loss { - EvacLoss::Memory { reason } => { - assert_eq!(reason, "source-dead-no-snapshot"); - } - other => panic!("expected Memory loss, got {other:?}"), - } - assert_eq!(receipt.new_sandbox_id, new_sandbox); - - // The recovery went through create() with the live manifest as - // the rootfs override — the fresh-kernel-boot shape. - let spec = target_be - .last_create_spec() - .expect("disk-only recovery must call create(), not restore()"); - assert_eq!(spec.rootfs_manifest, Some(live)); - - // PG was rebound through HostLost → Created. - let updated = meta.get_session(session_id).await.expect("session"); - assert_eq!(updated.status, SessionState::Created); - } - - /// Arm 3b (ADR 0028 Fix B): disk-only recoverable but no cold-boot - /// spec (image no longer enabled) → structural ColdBootUnavailable, - /// PG untouched. The resumer fail-fasts this to Idle instead of - /// burning its budget. - #[tokio::test] - async fn evac_dead_source_disk_only_without_spec_is_structural() { - let meta = sim_meta(); - let session = stage_hostlost_session( - &meta, - HostId::new(), - SandboxId::new(), - Some(fake_manifest(0xCAFE, 9)), - ) - .await; - let session_id = session.id; - - let (registry, _target_host, _target_be) = build_registry_with_target(meta.clone()).await; - - let result = evacuate_dead_source( - ®istry, - &(meta.clone() as Arc), - session.clone(), - None, - ready_spec(None), - None, - None, - SessionFence::unfenced(), - None, // #800: budget — tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await; - match &result { - Err(e @ EvacError::ColdBootUnavailable(_)) => assert!(e.is_structural()), - other => panic!("expected ColdBootUnavailable, got {other:?}"), - } - assert_eq!( - meta.get_session(session_id).await.expect("session").status, - SessionState::HostLost - ); - } - - /// ADR 0116 (#1212 review finding): a TRANSIENT materialization error - /// on the disk-only rung surfaces as `SpecResolution` — retryable, not - /// structural — so the caller's next tick re-derives it instead of - /// failing the session fast. - #[tokio::test] - async fn evac_dead_source_disk_only_spec_read_error_is_retryable() { - let meta = sim_meta(); - let live = fake_manifest(0xBEEF, 4); - let session = - stage_hostlost_session(&meta, HostId::new(), SandboxId::new(), Some(live)).await; - - let (registry, _target_host, _target_be) = build_registry_with_target(meta.clone()).await; - - let result = evacuate_dead_source( - ®istry, - &(meta.clone() as Arc), - session.clone(), - None, - std::future::ready(Err(crate::error::ApiError::Internal( - "runtime spec read failed (simulated PG blip)".into(), - ))), - None, - None, - SessionFence::unfenced(), - None, // #800: budget — tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await; - match &result { - Err(e @ EvacError::SpecResolution(_)) => assert!(!e.is_structural()), - other => panic!("expected SpecResolution, got {other:?}"), - } - } - - /// Arm 4: no snapshot, no live_disk → NoRecoverableState. Caller - /// (dead_host.rs) routes this to HostLost → Dead. - #[tokio::test] - async fn evac_dead_source_no_state_errors_no_recoverable() { - let meta = sim_meta(); - let session = stage_hostlost_session(&meta, HostId::new(), SandboxId::new(), None).await; - let session_id = session.id; - - let (registry, _target_host, _target_be) = build_registry_with_target(meta.clone()).await; - - let result = evacuate_dead_source( - ®istry, - &(meta.clone() as Arc), - session.clone(), - None, - ready_spec(None), - None, - None, - SessionFence::unfenced(), - None, // #800: budget — tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await; - assert!(matches!(result, Err(EvacError::NoRecoverableState))); - - // PG untouched at HostLost. - let updated = meta.get_session(session_id).await.expect("session"); - assert_eq!(updated.status, SessionState::HostLost); - } - - /// No registered target host → NoTargetAvailable. Caller logs + - /// may retry. - #[tokio::test] - async fn evac_dead_source_no_target_returns_no_target_available() { - let meta = sim_meta(); - let session = stage_hostlost_session( - &meta, - HostId::new(), - SandboxId::new(), - Some(fake_manifest(0x1, 1)), - ) - .await; - - // Empty registry — no hosts to pick. - let registry = Arc::new(HostRegistry::new(meta.clone())); - - let result = evacuate_dead_source( - ®istry, - &(meta.clone() as Arc), - session.clone(), - None, - ready_spec(Some(test_cold_boot_spec())), - None, - None, - SessionFence::unfenced(), - None, // #800: budget — tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await; - assert!(matches!(result, Err(EvacError::NoTargetAvailable(_)))); - } - - /// #800 (RESERVED evac placement): when the only survivor is - /// MEASURED-FULL for the session's budget, `evacuate_dead_source` - /// returns `NoCapacityQueue` (the resumer queues) instead of soft-binding - /// the full host and driving Σ reserved > allocatable. Non-vacuous: the - /// SAME host + a budget that FITS places normally, proving the gate fires - /// on real over-subscription, not always. Disk-only recovery keeps the - /// UFFD-capability gate out of the picture, isolating the capacity gate. - #[tokio::test] - async fn evac_reserved_queues_when_no_survivor_fits() { - let meta = sim_meta(); - let session = stage_hostlost_session( - &meta, - HostId::new(), - SandboxId::new(), - Some(fake_manifest(0x5, 1)), - ) - .await; - - // One survivor, measured with 1024 MiB allocatable + a backend. - let registry = Arc::new(HostRegistry::new(meta.clone())); - let survivor = HostId::new(); - let backend = Arc::new(FakeBackend::default()); - registry.register(survivor, backend.clone()); - add_measured_host(&meta, survivor, 1024).await; - - // Budget larger than allocatable → fits NO survivor → NoCapacityQueue. - let result = evacuate_dead_source( - ®istry, - &(meta.clone() as Arc), - session.clone(), - None, - ready_spec(Some(test_cold_boot_spec())), - None, - None, - SessionFence::unfenced(), - Some((8192, 4)), - chrono::Utc::now(), - ) - .await; - assert!( - matches!(result, Err(EvacError::NoCapacityQueue)), - "an over-budget evac must queue, not bind a measured-full host: {result:?}", - ); - - // Non-vacuity: a budget that FITS the SAME host places normally. - let new_sandbox = SandboxId::new(); - backend.set_restore_id(new_sandbox); - let receipt = evacuate_dead_source( - ®istry, - &(meta.clone() as Arc), - session.clone(), - None, - ready_spec(Some(test_cold_boot_spec())), - None, - None, - SessionFence::unfenced(), - Some((512, 1)), - chrono::Utc::now(), - ) - .await - .expect("a fitting budget must place"); - assert_eq!(receipt.new_host_id, survivor); - assert_eq!(receipt.new_sandbox_id, new_sandbox); - } - - /// pick_evac_disk_manifest semantics: same lineage → newer wins; - /// different lineage → snapshot wins; None handling. Cheap pure- - /// function test, mirrors the api/snapshot.rs effective_resume - /// tests. - #[test] - fn pick_disk_prefers_live_only_when_newer_same_lineage() { - let live = fake_manifest(0xAA, 5); - let snap = fake_manifest(0xAA, 3); - assert_eq!(pick_evac_disk_manifest(Some(live), Some(snap)), Some(live)); - } - - #[test] - fn pick_disk_falls_back_to_snapshot_on_different_lineage() { - let live = fake_manifest(0xAA, 99); - let snap = fake_manifest(0xBB, 1); - assert_eq!(pick_evac_disk_manifest(Some(live), Some(snap)), Some(snap),); - } - - #[test] - fn pick_disk_handles_either_none() { - let snap = fake_manifest(0xAA, 1); - assert_eq!(pick_evac_disk_manifest(None, Some(snap)), Some(snap)); - let live = fake_manifest(0xBB, 1); - assert_eq!(pick_evac_disk_manifest(Some(live), None), Some(live)); - assert_eq!(pick_evac_disk_manifest(None, None), None); - } -} diff --git a/crates/engram-coordinator/src/grpc_app/fleet.rs b/crates/engram-coordinator/src/grpc_app/fleet.rs index 2697ca924..57cfce754 100644 --- a/crates/engram-coordinator/src/grpc_app/fleet.rs +++ b/crates/engram-coordinator/src/grpc_app/fleet.rs @@ -92,30 +92,17 @@ impl app::fleet_service_server::FleetService for AppFleetService { let owner = parse_owner(&input.owner)?; let meta = &self.state.services.meta; let now = self.state.services.clock.now_utc(); - let requested = meta - .request_host_retirement(host_id, owner, &input.reason, now) - .await - .map_err(|e| Status::internal(e.to_string()))?; - if !requested { - // Idempotent on a host that is already retired; honest on the - // rest (unknown, dead, or cordoned by another owner). - let row = meta - .get_host(host_id) - .await - .map_err(|e| Status::internal(e.to_string()))? - .ok_or_else(|| Status::not_found("host not found"))?; - if row.status != engram_core::types::host::HostStatus::Retired { - let why = match row.cordon_owner { - Some(other) if other != owner => { - format!("host is cordoned by {}", other.as_str()) - } - _ => format!("host is {}", row.status.as_str()), - }; - return Err(Status::failed_precondition(why)); - } - } - // TODO(ADR 0123 B4): C2 plans durable teleports here. - let teleports_planned = 0; + // Idempotent on a retired host; FailedPrecondition on the rest + // (unknown, dead, or cordoned by another owner). + let plan = crate::api::admin::request_host_retirement_core( + &self.state, + host_id, + owner, + engram_core::types::teleport::TeleportReason::RetireHost, + ) + .await + .map_err(into_status)?; + let teleports_planned = plan.planned.len() as u32; meta.grant_host_retirement(host_id, now) .await .map_err(|e| Status::internal(e.to_string()))?; @@ -290,15 +277,9 @@ impl app::fleet_service_server::FleetService for AppFleetService { .map_err(into_status)?; Ok(Response::new(app::AdminDrainHostResponse { host_id: result.host_id.to_string(), - evacuating: result.evacuating.iter().map(|s| s.to_string()).collect(), - failures: result - .failures - .into_iter() - .map(|f| app::DrainFailure { - session_id: f.session_id.to_string(), - error: f.error, - }) - .collect(), + planned: result.planned.iter().map(ToString::to_string).collect(), + descended: result.descended.iter().map(ToString::to_string).collect(), + skipped: result.skipped, })) } @@ -430,19 +411,33 @@ impl app::fleet_service_server::FleetService for AppFleetService { Ok(Response::new(convert::flush_now_result_to_proto(result))) } - async fn evacuate_session( + async fn teleport_session( &self, - req: Request, - ) -> Result, Status> { + req: Request, + ) -> Result, Status> { self.auth.check(&req)?; - let id = parse_session_id(&req.get_ref().session_id)?; - // `target_host` is ignored per the proto comment (future override). - let result = crate::api::admin::evacuate_session_core(&self.state, id) + let input = req.into_inner(); + let id = parse_session_id(&input.session_id)?; + let target = input + .target_host + .map(|h| { + h.parse() + .map_err(|_| Status::invalid_argument("malformed target_host")) + }) + .transpose()?; + let row = crate::teleport::admit_for_rpc(&self.state, id, target) .await - .map_err(into_status)?; - Ok(Response::new(app::EvacuateSessionResponse { - session_id: result.session_id.to_string(), - status: result.status.to_string(), + .map_err(|e| match e { + crate::error::ApiError::Conflict(reason) if reason == "busy_lane" => { + Status::aborted(reason) + } + crate::error::ApiError::Conflict(reason) => Status::failed_precondition(reason), + other => into_status(other), + })?; + Ok(Response::new(app::TeleportSessionResponse { + teleport_id: row.id.to_string(), + kind: row.kind.as_str().into(), + dest_host_id: row.dest_host_id.to_string(), })) } diff --git a/crates/engram-coordinator/src/host_registry.rs b/crates/engram-coordinator/src/host_registry.rs index 1a5b4e018..901b808ce 100644 --- a/crates/engram-coordinator/src/host_registry.rs +++ b/crates/engram-coordinator/src/host_registry.rs @@ -580,6 +580,15 @@ impl HostClient for HostRegistry { backend.snapshot(id, fence).await } + async fn snapshot_hold( + &self, + id: SandboxId, + fence: SessionFence, + ) -> Result { + let (_, backend) = self.resolve_owner(id).await?; + backend.snapshot_hold(id, fence).await + } + async fn snapshot_begin( &self, id: SandboxId, diff --git a/crates/engram-coordinator/src/idle_evictor.rs b/crates/engram-coordinator/src/idle_evictor.rs index 195bbf8b2..dc539d227 100644 --- a/crates/engram-coordinator/src/idle_evictor.rs +++ b/crates/engram-coordinator/src/idle_evictor.rs @@ -424,13 +424,7 @@ async fn quarantine_reap_unevictable( /// `session_ops_one_running` index replaces the session lease) with /// durable step markers (`park_or_capture → mark_idle`). /// -/// `target_state = Idle` is the user-paused (manual /resume) shape; -/// `Evacuating` is the operator-drain shape where the `evac_resumer` -/// scanner drives `Evacuating → Created → Active` on a peer host — the -/// non-target-state code paths are IDENTICAL, so both flows share the -/// recoverability invariants (snapshot durable before destroy, PG flips -/// before the best-effort destroy so reconcile can't race the orphan -/// path). +/// This pipeline evicts to Idle. Teleport owns relocation capture and release. /// /// `nominated = true` (idle-detector / scanner / rung-descent ops) /// tightens the entry guard to `status == Evicting`: a rung-1/2 ascent @@ -447,14 +441,11 @@ pub(crate) async fn run_evict_pipeline( ) -> Result { let state = ctx.state; let session_id = ctx.op.session_id; - // The pipeline only knows about Idle and Evacuating as legal - // targets. Both share the "Active → captured-snapshot → suspended" - // semantic; any other target would skip half the steps and break - // the recovery invariants. Reject early with a clean error. - if !matches!(target_state, SessionState::Idle | SessionState::Evacuating) { + // Only eviction to Idle belongs to this pipeline. + if target_state != SessionState::Idle { return Err(EvictError::Meta(format!( "run_evict_pipeline: target {target_state:?} not supported \ - (only Idle and Evacuating)", + (only Idle)", ))); } @@ -1046,14 +1037,6 @@ pub(crate) async fn run_evict_pipeline( // heartbeat for this session because the reconcile pass keys // on Active status only. // - // Evacuating deliberately RETAINS the source binding. A successful - // destroy RPC is not sufficient ownership proof: the host may have - // durably accepted the verb while its teardown effect is still pending. - // The evac resumer re-destroys, probes the source, and clears this - // binding under its claim only after the sandbox is confirmed gone. - // Until then ADR 0090's coordinator truth continues to own the outgoing - // VM, so no replacement can be restored alongside it. - // // The transition's own facts (`snapshot_taken`, `evicted`, the final // `status_changed`) ride the SAME store transaction: the flip makes // the session immediately claimable, so post-commit `emit_fenced` @@ -1067,13 +1050,7 @@ pub(crate) async fn run_evict_pipeline( session_id, ctx.fence(), target_state, - // Post-#896: the Idle path detaches in the fused flip; Evacuating - // RETAINS the source binding until teardown is confirmed. - if target_state == SessionState::Idle { - BindingDisposition::Detach - } else { - BindingDisposition::Retain - }, + BindingDisposition::Detach, vec![ crate::state::SessionEvent::SnapshotTaken { snapshot_id: metadata.id, @@ -1355,7 +1332,7 @@ impl Default for EvictionScannerConfig { /// Spawn the eviction scanner as a background task. Caller holds the /// JoinHandle for the process lifetime; dropping aborts the loop. -/// Mirrors [`crate::evac_resumer::spawn`]. +/// Mirrors [`crate::teleport::spawn`]. /// /// The first sweep after coord startup is part of the deploy-recovery /// story: a row left `Evicting` with no op (see the module doc) gets a @@ -4181,11 +4158,11 @@ mod tests { .await .expect("drain"); assert_eq!( - resp.evacuating, + resp.descended, vec![session_id], "the parked session is drain work, not an empty success" ); - assert!(resp.failures.is_empty(), "failures: {:?}", resp.failures); + assert_eq!(resp.skipped, 0); { let m = meta.clone(); diff --git a/crates/engram-coordinator/src/lib.rs b/crates/engram-coordinator/src/lib.rs index 0c2e603c9..e4406ae52 100644 --- a/crates/engram-coordinator/src/lib.rs +++ b/crates/engram-coordinator/src/lib.rs @@ -22,8 +22,6 @@ pub mod cow_state; pub mod dead_host; pub mod enable_scanner; pub mod error; -pub mod evac_resumer; -pub mod evacuation; pub mod grpc_app; pub mod harness_catalog; pub mod harness_desync; @@ -33,7 +31,6 @@ pub mod idle_detector; pub mod idle_evictor; pub mod integration_ops; pub mod integrations; -pub mod live_migration; pub mod metrics; pub mod oauth; pub mod oauth_redirect; @@ -55,6 +52,7 @@ pub mod snapshot_blob_gc; mod span_parenting_tests; pub mod squashfs; pub mod state; +pub mod teleport; pub use config::CoordinatorConfig; pub use error::ApiError; @@ -208,14 +206,7 @@ pub async fn run_with_registry_and_local( // simulator. An expired binding lease (ADR 0116 A-D4) triggers // eviction within ~host_lease_ttl + poll_interval. let _dead_host = dead_host::spawn(dead_host::DeadHostConfig::default(), state.clone()); - // ADR 0018 commit 12c: the evac-resumer scanner picks up sessions - // marked Evacuating (by the admin /drain, /evacuate, or - // dead_host.rs) and drives Evacuating → Created → Active on a - // peer host via the shared resume primitives. Without it, - // sessions transitioned to Evacuating just sit there. Doesn't - // require its own PgPool — it goes through MetadataStore. - let _evac_resumer = - evac_resumer::spawn(evac_resumer::EvacResumerConfig::default(), state.clone()); + let _teleport = teleport::spawn(teleport::TeleportConfig::default(), state.clone()); // ADR 0048: the session queue scanner. Drives `queued` sessions to // placement (best-fit, per-fit-class FIFO) as capacity frees / the diff --git a/crates/engram-coordinator/src/live_migration.rs b/crates/engram-coordinator/src/live_migration.rs deleted file mode 100644 index ac1e8caa8..000000000 --- a/crates/engram-coordinator/src/live_migration.rs +++ /dev/null @@ -1,1819 +0,0 @@ -//! ADR 0045 C2: the POST-COPY live-teleport coordinator verb (the -//! clean-break replacement of C1's stop-and-copy; snapshot-rehome is -//! the only fallback). -//! -//! `migrate_session_live` under the session's op-log claim (ADR 0079, -//! kind = teleport) + the R8 host gate: -//! `migration_presetup` on the source (pre-pause: export identity + -//! the dest's restore package) → SPAWN the dest restore as a task (it -//! pre-stages netns/FC/handler concurrently with everything below; its -//! handler parks at the source's page server until the SEAL, and its -//! FC load gates on `state.bin` landing) → mark `Evacuating` (the -//! crash parachute) → `migration_capture_postcopy` (THE BLACKOUT: -//! pause → NBD drain → fork-v3 vmstate-only → pagemap seal) → await -//! the restore task → the `Committing` persist (one atomic -//! `rebind_session` UPDATE — the ownership oracle flips with it) → -//! emit `evacuating → active` (post-blackout; the prompt-hold keeps -//! "messages deliver" honest through the harness rebuild) → finalize -//! task: `migration_drain_wait` (the dest pulls every sealed chunk) → -//! commit (destroy) the source. Durability is NOT migration's job -//! (operator decision, 2026-06-12, superseding the opening bookend's -//! D-decision): the dest simply joins the periodic checkpoint cadence -//! like any resumed session — its first periodic capture is a safe -//! Full (no chain) — and until that lands, recovery rewinds to the -//! SOURCE's last periodic row (RPO ≤ cadence, the accepted model). -//! The old finalize took an immediate post-move Full here; that was a -//! guest-visible 2-4s pause seconds after landing, paid on every -//! single move to shave the rewind window — backwards, given the -//! goal is a teleport humans can't notice. -//! -//! Failure arms: presetup/dest-prep failures before the pause ⇒ -//! `Unsupported`/`Fatal`, session untouched (snapshot-rehome fallback). -//! Capture failure ⇒ best-effort resume-in-place + walk back to Active -//! (zero loss). Dest restore failing with the `postcopy-never-loaded` -//! marker (its FC load gate timed out — the dest PROVABLY never ran -//! the shipped state) ⇒ `migration_abort` un-pauses the source (zero -//! loss); any other/ambiguous restore failure ⇒ parachute (scanner -//! rehome from the durable row, or kill when none exists). PeerLost -//! mid-drain ⇒ the finalize destroys the poisoned dest and re-arms -//! `Evacuating` (rung-1 rewind). - -use engram_core::types::snapshot::MigrationSourceInfo; -use engram_core::types::BindingDisposition; -use engram_core::types::SessionState; -use engram_core::SandboxError; -use engram_core::{HostId, SessionId}; -use tracing::Instrument; - -use crate::state::{SessionEvent, SharedState}; - -#[derive(Debug)] -pub enum MigrateError { - /// Source or destination can't do a live move — fall back to - /// snapshot-rehome (the pre-C1 teleport). - Unsupported(String), - /// The move failed but the source was aborted back to Active — - /// downtime only, zero loss. - AbortedToSource(String), - /// The move failed in a state the parachute owns: the session is - /// `Evacuating` and the scanner will rehome from the last durable - /// checkpoint (loss ≤ one cadence). - Parachute(String), - Fatal(String), -} - -impl std::fmt::Display for MigrateError { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - match self { - Self::Unsupported(m) => write!(f, "live migration unsupported: {m}"), - Self::AbortedToSource(m) => write!(f, "live migration aborted to source: {m}"), - Self::Parachute(m) => write!(f, "live migration failed; scanner rehome armed: {m}"), - Self::Fatal(m) => write!(f, "live migration failed: {m}"), - } - } -} - -/// Feature gate: `ENGRAM_LIVE_TELEPORT=1`. Off ⇒ the teleport verb keeps -/// the snapshot-rehome path unconditionally. -pub fn live_teleport_enabled() -> bool { - std::env::var("ENGRAM_LIVE_TELEPORT") - .map(|v| v == "1") - .unwrap_or(false) -} - -/// ADR 0045 C2 (R8): at most ONE in-flight migration per host endpoint -/// — a consolidation wave would otherwise double P2P+GCS pressure on a -/// single source/dest. Process-local (coordinator pods are effectively -/// singular today; the per-session op claim already serializes -/// per-session). -static MIGRATION_GATE: std::sync::LazyLock> = - std::sync::LazyLock::new(dashmap::DashMap::new); - -/// Both-endpoint claim; `Drop` releases both. Holds source AND dest for -/// the whole move INCLUDING the finalize drain (the guard rides into -/// the finalize task). -struct MigrationGateGuard { - hosts: Vec, -} - -impl MigrationGateGuard { - fn claim(source: Option, dest: HostId, session: SessionId) -> Option { - let mut claimed = Vec::new(); - for host in source.into_iter().chain([dest]) { - // Compute the claim outcome and DROP the entry guard before - // any rollback `remove` — holding a shard lock across a - // `remove` of a host that hashes to the SAME shard - // self-deadlocks (DashMap shards are RwLocks; fewer shards - // on low-core CI made the collision a real hang). - let inserted = match MIGRATION_GATE.entry(host) { - dashmap::mapref::entry::Entry::Vacant(v) => { - v.insert(session); - true - } - dashmap::mapref::entry::Entry::Occupied(_) => false, - }; - if inserted { - claimed.push(host); - } else { - // Roll back partial claims — no entry guard held now. - for h in claimed { - MIGRATION_GATE.remove(&h); - } - return None; - } - } - Some(Self { hosts: claimed }) - } -} - -impl Drop for MigrationGateGuard { - fn drop(&mut self) { - for h in &self.hosts { - MIGRATION_GATE.remove(h); - } - } -} - -// ADR 0019 / telemetry restoration (#526): this verb is driven both from -// an admin request fan-out and (indirectly) from scanner-driven drain — -// neither reliably supplies a request span. An explicit root (carrying -// `session_id`/`target_host_id`) means the pipeline's own detached spawns -// below (dest restore, drain+commit finalize) have something real to -// `.instrument(Span::current())` onto instead of orphaning. -#[tracing::instrument(name = "live_migration.migrate_session_live", skip_all, fields(%session_id, %target_host_id))] -pub async fn migrate_session_live( - state: &SharedState, - session_id: SessionId, - target_host_id: HostId, -) -> Result<(), MigrateError> { - let clock = state.services.clock.clone(); - let t_total = clock.now_mono(); - // Serialize against resumes / evictions / sibling migrations. - let session = state - .services - .meta - .get_session(session_id) - .await - .map_err(|e| MigrateError::Fatal(format!("get_session: {e}")))?; - if session.status != SessionState::Active { - return Err(MigrateError::Fatal(format!( - "live migration needs an Active session (got {})", - session.status.as_str(), - ))); - } - let Some(sandbox_id) = session.sandbox_id else { - return Err(MigrateError::Fatal("no bound sandbox".into())); - }; - // ADR 0112 D7: swap-armed guests cannot post-copy teleport. The - // guard lives HOST-side in `migration_presetup` (the live sandbox - // spec is the authoritative swap source — an image row can drift - // after capture); its `InvalidSpec` surfaces here as `Unsupported` - // and callers fall back to snapshot-rehome, whose `capture_phase` - // runs the swap disarm and is already correct. - // ADR 0079: the op-log claim (kind = teleport) is the per-session - // exclusion — one running op per session. Claim-or-give-up; if this - // pod dies mid-move the reclaim sweep re-claims the row and the - // teleport verb arm terminally fails it (the ADR 0018 parachute / - // evac machinery owns recovery), freeing the lane. - let claim = match crate::session_ops::OpClaim::try_acquire( - state, - session_id, - engram_core::types::session_op::OpKind::Teleport, - serde_json::json!({ "target_host_id": target_host_id.to_string() }), - ) - .await - { - Ok(Some(c)) => c, - Ok(None) => { - return Err(MigrateError::Fatal( - "session is mid-resume/eviction/migration (an op owns it)".into(), - )) - } - Err(e) => return Err(MigrateError::Fatal(format!("op claim acquire: {e}"))), - }; - - // ADR 0079 (re-review finding #1): keep the claim's `heartbeat_at` - // fresh for the ENTIRE synchronous move body (presetup → concurrent - // restore-await → blackout → rebind → reactivate below). Without this - // beat the row's only liveness stamp is `try_acquire`'s, so a move - // whose body outlives `RECLAIM_STALE` (180s — a large VM over a slow - // inter-host link) is reclaimed out from under a PERFECTLY HEALTHY - // holder, which then keeps driving un-fenced migration RPCs while the - // successor's reclaim fails the row (double-drive). The manual-snapshot - // inline claim already beats its body this way (`api::snapshot`); the - // teleport body did not. Dropped explicitly right before the finalize - // task spawns (that task runs its own `claim.touch` beat across the - // drain); any early `?`/return arm drops it via RAII. - let move_heartbeat = claim.spawn_heartbeat("teleport-move"); - - // The destination must be takeable and the source addressable - // before we freeze anything. - let (_, dest_backend) = crate::placement::pick_specific_host( - state.services.meta.as_ref(), - &state.host_registry, - target_host_id, - session.host_id, - clock.now_utc(), - ) - .await - .map_err(|e| MigrateError::Fatal(format!("target host can't take the session: {e:?}")))?; - let source_addr = source_host_addr(state, session.host_id) - .await - .ok_or_else(|| { - MigrateError::Unsupported("source host has no advertised host_addr".into()) - })?; - - // The last durable checkpoint row: the metadata template AND the - // parachute's landing spot. OPTIONAL by operator decision - // (2026-06-11): a session with no durable row yet (younger than - // its first periodic checkpoint) teleports anyway — if the move - // fails mid-flight there is nothing to rehome from and the - // session is lost. Metadata falls back to the image's base - // snapshot row (aux_bundles must still pin, or a [git] image - // restores without its skills mounts). - let durable_row = state - .services - .meta - .latest_snapshot_for_session(session_id) - .await - .map_err(|e| MigrateError::Fatal(format!("latest snapshot: {e}")))?; - let base_row = match &durable_row { - Some(_) => None, - None => match state - .services - .meta - .get_enabled_image_any(&session.image) - .await - { - Ok(Some(img)) => match img.base_snapshot_id { - Some(id) => state.services.meta.get_snapshot(id).await.ok().flatten(), - None => None, - }, - _ => None, - }, - }; - if durable_row.is_none() { - tracing::warn!( - %session_id, - "live migration without a durable checkpoint row — a mid-move failure past the freeze CANNOT be rehomed (operator-accepted)", - ); - } - - // Resolve the source's backend handle ONCE, pre-freeze, and use it - // for capture, the failure-arm aborts, AND the post-rebind commit. - // Commit cannot route by sandbox id: step 4's - // `invalidate_sandbox(sandbox_id)` drops the old route on purpose - // (the old sandbox must stop serving exec), and the PG read-through - // finds nothing because the session row already points at the new - // sandbox — prod canary 5fa742b7 hit exactly this ("sandbox not - // found" on commit; the source stayed frozen until the export TTL - // destroyed it ~2 min later). - let source_backend = match state.host_registry.resolve_owner(sandbox_id).await { - Ok((_, backend)) => backend, - Err(e) => return Err(MigrateError::Fatal(format!("resolve source host: {e}"))), - }; - - // ---- R8 gate: one in-flight migration per host endpoint ---- - let Some(gate_guard) = MigrationGateGuard::claim(session.host_id, target_host_id, session_id) - else { - return Err(MigrateError::Fatal( - "another migration is in flight on the source or destination host (R8)".into(), - )); - }; - - // ---- 1. Presetup on the source (NO pause — the guest runs) ---- - let t_presetup = clock.now_mono(); - let presetup = match source_backend - .migration_presetup(sandbox_id, claim.fence()) - .await - { - Ok(p) => p, - Err(SandboxError::InvalidSpec(reason)) => { - return Err(MigrateError::Unsupported(reason)); - } - Err(e) => return Err(MigrateError::Fatal(format!("migration presetup: {e}"))), - }; - let presetup_ms = clock.now_mono().saturating_sub(t_presetup).as_millis(); - - // The page-server address: the source's advertised gRPC host with - // the presetup's peer port (default 9102). - let peer_addr = { - let hostpart = source_addr - .trim_start_matches("http://") - .trim_start_matches("https://"); - let host = hostpart - .rsplit_once(':') - .map(|(h, _)| h) - .unwrap_or(hostpart); - format!("{host}:{}", presetup.peer_port) - }; - - // ---- 2. Spawn the destination restore (pre-stage) ---- - // Runs CONCURRENTLY with the blackout below: netns + FC spawn + - // handler bring-up happen while the guest still runs; the handler - // parks at the page server for the SEAL and the FC load gates on - // `state.bin` (landed by the dest's fetch poller once the capture - // opens the export). No artifacts exist yet by design. - let base_memory_manifest = - crate::api::snapshot::base_memory_manifest_for_image(state, &session.image).await; - let metadata = engram_core::types::snapshot::SnapshotMetadata { - // The restore stages under a FRESH id (post-copy mints no - // capture-time snapshot id the dest could collide on). - id: engram_core::types::SnapshotId::new(), - size_bytes: durable_row.as_ref().map(|r| r.size_bytes).unwrap_or(0), - created_at: clock.now_utc(), - image_version: durable_row - .as_ref() - .map(|r| r.image_version.clone()) - .or_else(|| base_row.as_ref().map(|r| r.image_version.clone())) - .unwrap_or_else(|| { - session - .image - .rsplit(':') - .next() - .unwrap_or("unknown") - .to_string() - }), - base_memory_manifest, - migration_source: Some(MigrationSourceInfo { - export_id: presetup.export_id.clone(), - source_addr: source_addr.clone(), - memory_manifest_json: presetup.memory_manifest_json.clone(), - disk_manifest_json: Vec::new(), - memory_manifest_ref: presetup.memory_manifest_ref, - disk_manifest_ref: presetup - .disk_manifest_ref - .unwrap_or_else(engram_core::types::manifest::ManifestRef::new), - new_memory_chunk_hashes: Vec::new(), - new_disk_chunk_hashes: Vec::new(), - hot_chunks: presetup.hot_chunks.clone(), - post_copy: true, - peer_addr: Some(peer_addr), - peer_token: Some(presetup.peer_token.clone()), - sidecar_json: presetup.sidecar_json.clone(), - }), - disk_manifest: presetup.disk_manifest_ref, - memory_manifest: Some(presetup.memory_manifest_ref), - source_sandbox_id: None, - state_blob_key: None, - sidecar_blob_key: None, - rootfs_blob_key: None, - working_set_blob_key: None, - aux_bundles: durable_row - .as_ref() - .map(|r| r.aux_bundles.clone()) - .or_else(|| base_row.as_ref().map(|r| r.aux_bundles.clone())) - .unwrap_or_default(), - // Issue #529: restore-side reconstruction, not a fresh capture — - // no pause instant to carry. - paused_at: None, - // ADR 0095: peer-fill hints are for ordinary resumes; a teleport dest pulls via its migration export. - peer_hints: Vec::new(), - }; - // ADR 0019 / telemetry restoration (#526): `dest.restore` makes a - // gRPC call to the target host-agent; the `TraceparentInjector` - // interceptor propagates whatever span is current at call time onto - // the wire. Detaching via bare `tokio::spawn` would send an empty - // traceparent and orphan the host-side restore spans from this - // migration trace. - let restore_task = { - let dest = dest_backend.clone(); - let restore_span = tracing::Span::current(); - // ADR 0079: the teleport claim's epoch fences the dest restore. - let fence = claim.fence(); - tokio::spawn(async move { dest.restore(metadata, fence).await }.instrument(restore_span)) - }; - - // ---- 3. Arm the parachute ---- - // From here until the rebind, a coordinator death leaves the - // session Evacuating: lease expiry → scanner snapshot-rehome from - // `durable_row`. The frozen source self-cleans via the export TTL - // (whose yes⇒un-pause arm is FORBIDDEN once state.bin shipped). - if let Err(e) = state - .services - .meta - .transition_session( - session_id, - SessionState::Evacuating, - BindingDisposition::Retain, - ) - .await - { - restore_task.abort(); - return Err(MigrateError::Fatal(format!("transition: {e}"))); - } - if let Err(e) = state - .services - .meta - .set_teleport_target(session_id, Some(target_host_id)) - .await - { - tracing::warn!(%session_id, error = %e, - "set_teleport_target failed; parachute would fall back to any-peer"); - } - // Observer-facing truth: the session leaves `active` as the - // blackout begins. - let _ = state - .emit( - session_id, - SessionEvent::StatusChanged { - from: SessionState::Active, - to: SessionState::Evacuating, - at: clock.now_utc(), - }, - ) - .await; - - // ---- 4. THE BLACKOUT: vmstate-only capture + pagemap seal ---- - let t_blackout = clock.now_mono(); - let capture = match source_backend - .migration_capture_postcopy(sandbox_id, &presetup.export_id, claim.fence()) - .await - { - Ok(c) => c, - Err(e) => { - // Nothing shipped; the guest may or may not be paused - // depending on where capture failed — resume is the - // conservative un-freeze (idempotent enough: resuming a - // running VM is a benign FC error). - restore_task.abort(); - let _ = source_backend.resume(sandbox_id, claim.fence()).await; - let _ = state - .services - .meta - .set_teleport_target(session_id, None) - .await; - if walk_back_to_active(state, session_id).await { - return Err(MigrateError::AbortedToSource(format!( - "post-copy capture: {e}" - ))); - } - return Err(parachute_or_kill( - state, - session_id, - durable_row.is_some(), - format!("post-copy capture: {e}"), - ) - .await); - } - }; - metrics::histogram!(crate::metrics::MIGRATION_LEG_SECONDS, "leg" => "capture_postcopy") - .record(clock.now_mono().saturating_sub(t_blackout).as_secs_f64()); - - // ---- 5. Await the destination (load + resume) ---- - let t_restore = clock.now_mono(); - let new_sandbox_id = match restore_task.await { - Ok(Ok(id)) => id, - Ok(Err(e)) => { - let _ = state - .services - .meta - .set_teleport_target(session_id, None) - .await; - // The load-gate marker proves the dest never ran the - // shipped state — un-pausing the source is zero-loss - // sound. Anything else is ambiguous: parachute. - if e.to_string().contains("postcopy-never-loaded") { - let abort_ok = source_backend - .migration_abort(sandbox_id, &presetup.export_id, claim.fence()) - .await - .is_ok(); - if abort_ok && walk_back_to_active(state, session_id).await { - metrics::counter!(crate::metrics::MIGRATION_TOTAL, "outcome" => "aborted_to_source") - .increment(1); - return Err(MigrateError::AbortedToSource(format!("dest restore: {e}"))); - } - } - return Err(parachute_or_kill( - state, - session_id, - durable_row.is_some(), - format!("dest restore: {e}"), - ) - .await); - } - Err(join_err) => { - let _ = state - .services - .meta - .set_teleport_target(session_id, None) - .await; - return Err(parachute_or_kill( - state, - session_id, - durable_row.is_some(), - format!("dest restore task: {join_err}"), - ) - .await); - } - }; - let restore_ms = clock.now_mono().saturating_sub(t_restore).as_millis(); - // The guest is RUNNING on the dest from here (FC resumed inside - // the load) — this is where the guest-observed blackout ends. The - // rebind + harness rebuild below happen while the guest executes, - // so folding them into `blackout_ms` (the old shape) overstated - // the user-facing gap by the `finish_resume_to_active` wall. - let blackout_wall_ms = clock.now_mono().saturating_sub(t_blackout).as_millis() as u64; - let t_reactivate = clock.now_mono(); - - // ---- 6. The Committing persist + reactivate ---- - state.host_registry.invalidate_sandbox(sandbox_id); - state - .host_registry - .record_sandbox_owner(new_sandbox_id, target_host_id); - let rebind = async { - // ONE atomic UPDATE: the ownership oracle (`sandbox_ownership`) - // flips with it — the post-copy ownership transfer point. The - // source's TTL answer goes `false` from here. - // - // Issue #211: guard the rebind on the row still being the - // `Evacuating` row bound to the SOURCE sandbox we're migrating - // off. A reconcile strike racing the copy can flip the row - // and/or clear its sandbox; a blind rebind would re-bind the new - // live VM onto that row, the ownership oracle would answer - // `owned = true`, and the orphan reap would never fire. On - // `Conflict` we drop into the error arm below, which destroys - // `new_sandbox_id` and parachutes. - // - // ADR 0079 note: this stays `rebind_session_guarded` (state-list - // guard) rather than `fenced_assign_sandbox` even though DELETE - // now rides the op log (a destroy op queues behind this claim): - // the fleet detectors (reconcile's strike-out, dead_host) still - // write session rows WITHOUT bumping `current_epoch`, so the - // epoch predicate alone cannot see their flips. The state guard - // collapses into the fence once the detectors' writes become - // fenced enqueues (the ADR's "NOT deleted" list). - let epoch = state - .services - .meta - .rebind_session_guarded( - session_id, - target_host_id, - new_sandbox_id, - Some(Some(sandbox_id)), - &[SessionState::Evacuating], - ) - .await - .map_err(|e| format!("rebind: {e}"))?; - state - .services - .meta - .transition_session( - session_id, - SessionState::Created, - BindingDisposition::Retain, - ) - .await - .map_err(|e| format!("to Created: {e}"))?; - Ok::(epoch) - } - .await; - let epoch = match rebind { - Ok(epoch) => epoch, - Err(e) => { - let _ = dest_backend.destroy(new_sandbox_id, claim.fence()).await; - let _ = state - .services - .meta - .set_teleport_target(session_id, None) - .await; - return Err(parachute_or_kill(state, session_id, durable_row.is_some(), e).await); - } - }; - let bind_result = - crate::api::snapshot::bind_harness_generation(state, session_id, new_sandbox_id, epoch) - .await; - // CASE 1 (issue #209): the teleport_target pin's ONLY job is to aim - // the parachute at the dest while the move is in flight. The rebind - // above committed the ownership flip — the dest is now the durable - // owner — so the pin has done its job and MUST be cleared here, - // BEFORE the fallible refresh / finish_resume_to_active below. The - // pin is PG-durable (read back at evac_resumer's get_teleport_target, - // honored strictly as require_host), so leaking it on an early `?` - // return pins every future rehome of this session to this host for - // the process lifetime. Clearing it unconditionally on this exit - // path closes that leak; if the move rewinds further down, the - // parachute re-arms its own targeting. - let _ = state - .services - .meta - .set_teleport_target(session_id, None) - .await; - // Always schedule source finalization after the ownership transfer. - // Return the attach error after that task owns the drain. - let attach_result = async { - bind_result?; - let session_refreshed = state.services.meta.get_session(session_id).await?; - let plan = crate::boot_materializer::materialize_snapshot_resume( - state, - &session_refreshed, - new_sandbox_id, - epoch, - ) - .await?; - crate::api::snapshot::finish_resume_to_active( - state, - &session_refreshed, - new_sandbox_id, - plan, - claim.fence(), - ) - .await - } - .await; - - metrics::histogram!(crate::metrics::MIGRATION_LEG_SECONDS, "leg" => "presetup") - .record(presetup_ms as f64 / 1000.0); - metrics::histogram!(crate::metrics::MIGRATION_LEG_SECONDS, "leg" => "restore") - .record(restore_ms as f64 / 1000.0); - metrics::histogram!(crate::metrics::MIGRATION_LEG_SECONDS, "leg" => "total") - .record(clock.now_mono().saturating_sub(t_total).as_secs_f64()); - // PR 10 blackout decomposition: the legs that actually cost, so an - // optimization targets the real hot leg. Source-measured (under the - // freeze); the coordinator-side `blackout_ms` is the wall including - // the round trip. - metrics::histogram!(crate::metrics::MIGRATION_LEG_SECONDS, "leg" => "blackout_pause") - .record(capture.pause_ms as f64 / 1000.0); - metrics::histogram!(crate::metrics::MIGRATION_LEG_SECONDS, "leg" => "blackout_disk_drain") - .record(capture.disk_drain_ms as f64 / 1000.0); - metrics::histogram!(crate::metrics::MIGRATION_LEG_SECONDS, "leg" => "blackout_vmstate") - .record(capture.vmstate_ms as f64 / 1000.0); - metrics::histogram!(crate::metrics::MIGRATION_LEG_SECONDS, "leg" => "blackout_scan") - .record(capture.scan_ms as f64 / 1000.0); - metrics::counter!(crate::metrics::MIGRATION_TOTAL, "outcome" => "migrated_postcopy") - .increment(1); - tracing::info!( - %session_id, - old_sandbox = %sandbox_id, - new_sandbox = %new_sandbox_id, - target_host = %target_host_id, - presetup_ms, - sealed_chunks = capture.sealed_chunks, - total_chunks = capture.total_chunks, - sealed_disk_chunks = capture.sealed_disk_chunks, - // Blackout decomposition (source-measured, under the freeze): - // pause + disk_drain + vmstate + scan ≈ the host-side blackout; - // `blackout_ms` is the coordinator wall from capture start to - // the dest restore returning (guest running) — the closest - // coordinator-side proxy for the guest-observed gap. - // `reactivate_ms` (rebind + emits + harness rebuild) runs - // while the guest already executes. - blackout_pause_ms = capture.pause_ms, - blackout_disk_drain_ms = capture.disk_drain_ms, - blackout_vmstate_ms = capture.vmstate_ms, - scan_ms = capture.scan_ms, - blackout_ms = blackout_wall_ms, - restore_await_ms = restore_ms, - reactivate_ms = clock.now_mono().saturating_sub(t_reactivate).as_millis() as u64, - total_ms = clock.now_mono().saturating_sub(t_total).as_millis(), - "post-copy live teleport landed (ADR 0045 C2); drain + durability finalizing", - ); - - // ---- 7. Finalize: drain → commit source → Full checkpoint → row ---- - // The op claim + the R8 gate ride into the task. The dest keeps - // serving the user throughout; the SOURCE stays alive as a page - // server until DrainDone. - // - // ADR 0019 / telemetry restoration (#526): re-parent onto the - // migration span so the finalize (drain + source commit) stitches - // under the same trace instead of exporting as an orphaned root. - let state2 = state.clone(); - let export_id = presetup.export_id.clone(); - let finalize_span = tracing::Span::current(); - // The synchronous body is done; the finalize task keeps the claim alive - // via its own `claim.touch` beat across the drain, so hand liveness off - // by dropping the body heartbeat here (avoids two beats racing on the - // same row — harmless, but tidy). - drop(move_heartbeat); - tokio::spawn( - async move { - let claim = claim; - let _gate_guard = gate_guard; - - // 7a. The drain: every sealed chunk lands on the dest. Wrapped in - // a bounded-backoff retry (CASE 2, issue #209): a TRANSIENT - // transport Err on `migration_drain_wait` (a dest host-agent pod - // roll, a network blip) used to hit a bare `return` that left the - // session Active on the dest while the source's export TTL fired - // — the ownership probe then answers owned=false and the host - // DESTROYS the frozen source's page server, which the dest may - // still need, poisoning it. Retry the RPC a few times; only on - // PERSISTENT failure take the SAME rung-1 rewind as PeerLost - // (destroy dest, re-arm Evacuating, parachute_or_kill) — never - // leave an Active session whose page server gets reaped. - const DRAIN_RETRY_BUDGET: u32 = 5; - // Base backoff (doubled each attempt: 2s, 4s, 8s, …). Overridable - // via `ENGRAM_MIGRATION_DRAIN_RETRY_BASE_MS` for ops tuning and so - // the regression test can drive the exhaustion path without a - // ~30s real wait. - let drain_retry_base = std::time::Duration::from_millis( - std::env::var("ENGRAM_MIGRATION_DRAIN_RETRY_BASE_MS") - .ok() - .and_then(|v| v.parse::().ok()) - .unwrap_or(2000), - ); - let clock = state2.services.clock.clone(); - let t_drain = clock.now_mono(); - let mut last_err: Option = None; - let drain_outcome = 'retry: { - for attempt in 0..DRAIN_RETRY_BUDGET { - // Refresh the op claim's heartbeat across attempts the - // same way the wait below does, so a multi-attempt retry - // isn't reclaimed out from under us (20s ≪ the executor's - // 180s staleness bound, `RECLAIM_STALE`) and never stomps a - // session a successor re-claimed. - let mut touch = tokio::time::interval(std::time::Duration::from_secs(20)); - touch.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); - touch.tick().await; - - let drain = state2.services.host.migration_drain_wait(new_sandbox_id); - tokio::pin!(drain); - let res = loop { - tokio::select! { - res = &mut drain => break res, - _ = touch.tick() => { - match claim.touch("finalize").await { - crate::session_ops::OpTouch::Held => {} - // A transport blip on the touch is NOT loss - // (CASE: claim touch). Keep waiting; the - // reclaim sweep still backstops a truly dead - // holder, and the drain RPC remains in flight. - crate::session_ops::OpTouch::TransientError(e) => { - tracing::warn!(%session_id, error = %e, - "post-copy finalize: claim touch transport error — retrying, not abandoning"); - } - crate::session_ops::OpTouch::Lost => { - tracing::error!(%session_id, - "post-copy finalize: op claim fenced mid-drain (successor re-claimed)"); - return; - } - } - } - } - }; - match res { - Ok(outcome) => break 'retry Ok(outcome), - Err(e) => { - tracing::warn!(%session_id, error = %e, attempt = attempt + 1, - budget = DRAIN_RETRY_BUDGET, - "post-copy drain_wait transport error; retrying with backoff"); - last_err = Some(e); - // Exponential backoff (2s, 4s, 8s, …) — bounded by - // the budget so the whole retry stays well inside - // the source export TTL when started promptly. - if attempt + 1 < DRAIN_RETRY_BUDGET { - clock.sleep(drain_retry_base * (1 << attempt)).await; - } - } - } - } - // Budget exhausted: fall through to the rewind with the last - // transport error as the detail. - break 'retry Err(last_err - .map(|e| e.to_string()) - .unwrap_or_else(|| "drain_wait exhausted retries".into())); - }; - - // Normalize: a PeerLost outcome and a persistent transport Err - // both demand the rung-1 rewind. Map both into a single rewind - // detail so they share one arm (the page server is doomed either - // way once we stop owning the drain). - let rewind_detail: Option = match drain_outcome { - Ok(engram_core::types::snapshot::DrainOutcome::Done { - pulled, - alt_sourced, - zero_chunks, - ms, - }) => { - metrics::histogram!(crate::metrics::MIGRATION_LEG_SECONDS, "leg" => "drain") - .record(clock.now_mono().saturating_sub(t_drain).as_secs_f64()); - tracing::info!( - %session_id, pulled, alt_sourced, zero_chunks, ms, - "post-copy drain complete; releasing the source", - ); - None - } - Ok(engram_core::types::snapshot::DrainOutcome::PeerLost { remaining, detail }) => { - tracing::error!(%session_id, remaining, %detail, - "post-copy drain lost its peer — rewinding the whole VM (rung-1)"); - Some(format!("peer lost mid-drain: {detail}")) - } - Err(detail) => { - tracing::error!(%session_id, %detail, - "post-copy drain_wait failed after retries — rewinding the whole VM (rung-1)"); - Some(format!("drain transport error after retries: {detail}")) - } - }; - if let Some(detail) = rewind_detail { - // Rung-1 rewind: the dest holds unfillable state (or we can no - // longer drive the drain). Destroy it, re-arm Evacuating, and - // let the scanner rehome from the durable row (or kill when - // none exists). The frozen source: ownership now answers - // `false` (rebind landed), so its TTL destroys it. - metrics::counter!(crate::metrics::MIGRATION_TOTAL, "outcome" => "peer_lost_rewind") - .increment(1); - let _ = state2 - .services - .host - .destroy(new_sandbox_id, claim.fence()) - .await; - state2.host_registry.invalidate_sandbox(new_sandbox_id); - if state2 - .services - .meta - .transition_session( - session_id, - SessionState::Evacuating, - BindingDisposition::Retain, - ) - .await - .is_ok() - { - let _ = state2 - .emit( - session_id, - SessionEvent::StatusChanged { - from: SessionState::Active, - to: SessionState::Evacuating, - at: clock.now_utc(), - }, - ) - .await; - } - let _ = - parachute_or_kill(&state2, session_id, durable_row.is_some(), detail.clone()) - .await; - claim - .finish( - engram_core::types::session_op::OpState::Failed, - Some(&detail), - ) - .await; - return; - } - - // 7b. Release the source (destroy; the export retires with it). - if let Err(e) = source_backend - .migration_commit(sandbox_id, &export_id, claim.fence()) - .await - { - tracing::warn!(%session_id, %sandbox_id, error = %e, - "post-copy source commit failed; export TTL will clean up"); - } - - // That's it — durability is deliberately NOT migration's job - // (operator decision 2026-06-12; supersedes the bookend's - // "memory durability catch-up" arm). The dest joined the - // periodic checkpoint cadence at restore like any resumed - // session; its first periodic capture is a safe Full (no - // chain), and until that lands recovery rewinds to the - // source's last periodic row — RPO ≤ cadence, the accepted - // model. The immediate post-move Full that used to live here - // cost a guest-visible 2-4s pause seconds after EVERY move - // (prod 962011bf was its pathological form) to shave a rewind - // window nobody asked to shave; the teleport's job ends when - // the guest is live on the dest and the source is released. - tracing::info!(%session_id, - "post-copy migration finalized (source released; durability rides the periodic cadence)"); - claim - .finish(engram_core::types::session_op::OpState::Done, None) - .await; - } - .instrument(finalize_span), - ); - match attach_result.map_err(|e| MigrateError::Fatal(e.to_string()))? { - crate::api::snapshot::FinishResumeOutcome::Active => {} - crate::api::snapshot::FinishResumeOutcome::CreatedHarnessFailed { message, .. } => { - return Err(MigrateError::Fatal(message)); - } - } - Ok(()) -} - -/// The source host-agent's gRPC address. The heartbeat-warmed pool is -/// authoritative — `hosts.host_addr` in PG is written only at REGISTER, -/// and a host-agent pod that reattaches after a roll doesn't -/// re-register, leaving the PG row pointing at the PREVIOUS pod -/// generation (the prod canary's dest dialed a dead pod-network IP -/// exactly this way). PG is the fallback for a host the pool hasn't -/// warmed since the coordinator's own restart. -async fn source_host_addr(state: &SharedState, host_id: Option) -> Option { - let host_id = host_id?; - if let Some(addr) = state.services.host_pool.current_addr(host_id) { - return Some(addr); - } - let hosts = state.services.meta.list_active_hosts().await.ok()?; - hosts.into_iter().find(|h| h.id == host_id)?.host_addr -} - -/// Evacuating → Created → Active without relocating (the source VM was -/// aborted back in place; its bindings never changed). Returns false if -/// either transition is refused (the parachute then owns recovery). -/// The parachute's landing depends on whether a durable checkpoint row -/// exists. With one, leave the session `Evacuating` — the scanner -/// rehomes from the row (rung-1 semantics, loss ≤ cadence). WITHOUT -/// one there is nothing to rehome from: leaving `Evacuating` strands a -/// zombie the scanner grinds on forever, so kill the session outright -/// (operator decision, 2026-06-11 — the same call that dropped the -/// first-move row gate). -async fn parachute_or_kill( - state: &SharedState, - session_id: SessionId, - has_durable_row: bool, - msg: String, -) -> MigrateError { - if has_durable_row { - return MigrateError::Parachute(msg); - } - tracing::error!( - %session_id, - error = %msg, - "live migration failed past the freeze with NO durable checkpoint \ - row — killing the session (nothing to rehome from)", - ); - if let Err(e) = state - .services - .meta - .transition_session(session_id, SessionState::Failed, BindingDisposition::Detach) - .await - { - tracing::warn!(%session_id, error = %e, "kill-on-parachute: transition to Failed failed"); - } - let _ = state - .emit( - session_id, - SessionEvent::StatusChanged { - from: SessionState::Evacuating, - to: SessionState::Failed, - at: state.services.clock.now_utc(), - }, - ) - .await; - MigrateError::Fatal(format!( - "session lost (no durable checkpoint to rehome from): {msg}" - )) -} - -async fn walk_back_to_active(state: &SharedState, session_id: SessionId) -> bool { - for target in [SessionState::Created, SessionState::Active] { - if let Err(e) = state - .services - .meta - .transition_session(session_id, target, BindingDisposition::Retain) - .await - { - tracing::warn!(%session_id, ?target, error = %e, - "live migration walk-back transition refused"); - return false; - } - } - true -} - -#[cfg(test)] -// tests drive a live system; wall clock/OS entropy here is input, not a decision source (ADR 0098 D1) -#[allow(clippy::disallowed_methods)] -mod tests { - use super::*; - use crate::config::CoordinatorConfig; - use crate::host_registry::HostRegistry; - use crate::state::tests::MiniMeta; - use crate::state::AppState; - use crate::Services; - use engram_core::traits::MetadataStore; - use engram_core::types::session::{Session, SessionMode}; - use engram_core::SandboxId; - use std::sync::Arc; - - fn active_session() -> Session { - Session { - id: SessionId::new(), - status: SessionState::Active, - host_id: Some(HostId::new()), - sandbox_id: Some(SandboxId::new()), - image: "test/repo:live-migrate".into(), - // No harness and no image bundle in this fixture: dev-VM shape. - mode: SessionMode::DevVm, - created_at: chrono::Utc::now(), - last_active_at: chrono::Utc::now(), - last_event_at: None, - live_disk_manifest: None, - park_rung: 0, - parked_at: None, - suggested_title: None, - } - } - - fn build_state(session: Session) -> (SharedState, Arc, HostId) { - let tmp = std::env::temp_dir().join(format!("live-migrate-test-{}", session.id)); - std::fs::create_dir_all(&tmp).unwrap(); - let meta = Arc::new(MiniMeta::new(session)); - let host_registry = Arc::new(HostRegistry::new( - meta.clone() as Arc - )); - // A registered, ready target host (ProcessBackend-backed local - // client: every migration_* trait method is the default - // InvalidSpec — the pre-C1 host shape). - let target = HostId::new(); - let backend: Arc = Arc::new( - engram_sandbox_process::ProcessBackend::new(tmp.join("sandboxes")), - ); - host_registry.register( - target, - Arc::new(engram_host_agent::host_client::LocalHostClient::with_noop_hub(backend)), - ); - // ADR 0047: placement reads host rows — stage the target as a - // schedulable host in the mock store. - meta.add_ready_host(target); - let services = Services { - meta: meta.clone(), - host: host_registry.clone() as Arc, - secrets: Arc::new(engram_secrets_dev::InMemorySecretStore::new()), - kek: Arc::new(engram_crypto::EnvVarKeyProvider::from_bytes( - [0u8; 32], "test:v1", - )), - oci: Arc::new(engram_oci::OciClient::new(Arc::new( - engram_oci::AnonymousResolver, - ))), - auth_resolver: Arc::new(engram_oci::AnonymousResolver), - blob: Arc::new(engram_storage_local::LocalBlobStorage::new( - tmp.join("blobs"), - )), - chunk_store: engram_chunk_store::ChunkStore::new(Arc::new( - engram_storage_local::LocalBlobStorage::new(tmp.join("blobs")), - )), - host_pool: Arc::new(engram_protocol::grpc_pool::GrpcHostPool::new()), - materialize_dir: None, - clock: Arc::new(engram_core::traits::SystemClock::new()), - entropy: Arc::new(engram_core::traits::OsEntropy), - }; - let cfg = CoordinatorConfig { - local_path: tmp, - ..CoordinatorConfig::default() - }; - ( - Arc::new(AppState::new_with_registry(cfg, services, host_registry)), - meta, - target, - ) - } - - /// The fallback contract: a fleet that can't do a live move (no - /// durable row / no host_addr / pre-C1 capture) yields - /// `Unsupported` — never a frozen guest, never a state change — - /// and the caller falls back to snapshot-rehome. The session must - /// be left EXACTLY as found, lease released. - #[tokio::test] - async fn unsupported_fleet_falls_back_without_touching_the_session() { - let session = active_session(); - let session_id = session.id; - let (state, meta, target) = build_state(session); - - let err = migrate_session_live(&state, session_id, target) - .await - .expect_err("pre-C1 fleet must be unsupported"); - assert!( - matches!(err, MigrateError::Unsupported(_)), - "got {err:?} — only Unsupported triggers the snapshot-rehome fallback", - ); - let after = meta.get_session(session_id).await.unwrap(); - assert_eq!(after.status, SessionState::Active, "session untouched"); - // The op claim releases (Drop spawns a detached finish — poll - // briefly): the session's op lane must free for a follow-up. - let mut released = false; - for _ in 0..40 { - if meta.ops.running_for(session_id).is_none() { - released = true; - break; - } - tokio::time::sleep(std::time::Duration::from_millis(50)).await; - } - assert!(released, "the op claim must release after the fallback"); - } - - /// The dest-failure arm: capture succeeds (the source is frozen), - /// the destination restore fails — the verb ABORTS the source back - /// (un-pause in place) and walks the session to Active. Zero loss, - /// AbortedToSource posture. - #[tokio::test] - async fn dest_restore_failure_aborts_to_source_and_walks_back_to_active() { - use engram_core::types::snapshot::SnapshotMetadata; - use std::sync::atomic::{AtomicBool, Ordering}; - - struct MigratableFlakyDest { - abort_called: Arc, - } - #[async_trait::async_trait] - impl engram_core::traits::SandboxBackend for MigratableFlakyDest { - async fn create( - &self, - _: engram_core::types::sandbox::SandboxSpec, - ) -> Result { - Ok(SandboxId::new()) - } - async fn destroy(&self, _: SandboxId) -> Result<(), engram_core::SandboxError> { - Ok(()) - } - async fn list(&self) -> Result, engram_core::SandboxError> { - Ok(Vec::new()) - } - async fn exec_stream( - &self, - _: SandboxId, - _: engram_core::types::sandbox::ExecRequest, - ) -> Result - { - Err(engram_core::SandboxError::NotFound) - } - async fn snapshot( - &self, - _: SandboxId, - ) -> Result { - Err(engram_core::SandboxError::NotFound) - } - async fn migration_presetup( - &self, - _: SandboxId, - ) -> Result - { - Ok(engram_core::types::snapshot::MigrationPresetupOut { - export_id: "test-export".into(), - peer_token: "test-token".into(), - peer_port: 9102, - sidecar_json: b"{}".to_vec(), - memory_manifest_json: b"{}".to_vec(), - memory_manifest_ref: engram_core::types::manifest::ManifestRef::new(), - disk_manifest_ref: None, - hot_chunks: vec![], - }) - } - async fn migration_capture_postcopy( - &self, - _: SandboxId, - _: &str, - ) -> Result - { - Ok(engram_core::types::snapshot::PostCopyCaptureOut { - sealed_chunks: 3, - total_chunks: 16, - pause_ms: 1, - disk_drain_ms: 1, - vmstate_ms: 1, - scan_ms: 1, - sealed_disk_chunks: 0, - paused_at_unix_ms: 0, - }) - } - async fn resume(&self, _: SandboxId) -> Result<(), engram_core::SandboxError> { - Ok(()) - } - async fn migration_abort( - &self, - _: SandboxId, - _: &str, - ) -> Result<(), engram_core::SandboxError> { - self.abort_called.store(true, Ordering::SeqCst); - Ok(()) - } - async fn restore( - &self, - _: SnapshotMetadata, - ) -> Result { - // The load-gate marker: the dest provably never ran the - // shipped state — the coordinator's zero-loss abort arm. - Err(engram_core::SandboxError::Snapshot( - "postcopy-never-loaded: injected dest failure".into(), - )) - } - fn snapshot_path_for(&self, _: engram_core::types::SnapshotId) -> std::path::PathBuf { - std::path::PathBuf::from("/nonexistent") - } - } - - let session = active_session(); - let session_id = session.id; - let source_host = session.host_id.expect("source host"); - let (state, meta, target) = build_state(session); - // Replace the registry's target backend with the migratable - // flaky one — same host id, capture-capable, restore-failing. - let abort_called = Arc::new(AtomicBool::new(false)); - let flaky: Arc = Arc::new(MigratableFlakyDest { - abort_called: abort_called.clone(), - }); - state.host_registry.register( - target, - Arc::new(engram_host_agent::host_client::LocalHostClient::with_noop_hub(flaky.clone())), - ); - // The SOURCE is resolved through services.host (the registry) by - // sandbox owner — record the source sandbox's owner as the same - // flaky backend (it serves capture + abort). - state - .host_registry - .record_sandbox_owner(meta.session.lock().sandbox_id.unwrap(), target); - // Source host row with an addr + a durable checkpoint row. - meta.hosts - .lock() - .push(engram_core::types::host::HostRecord { - id: source_host, - hostname: "src".into(), - cloud_metadata: engram_core::types::host::HostMetadata::default(), - capacity: engram_core::types::host::HostCapacity { - total_gb: 100, - used_gb: 10, - total_mib: 65_536, - used_mib: 0, - running_sandboxes: 0, - }, - utilization: Default::default(), - status: engram_core::types::host::HostStatus::Ready, - last_heartbeat_at: chrono::Utc::now(), - host_addr: Some("http://127.0.0.1:1".into()), - ready_images: Vec::new(), - current_bundles: Vec::new(), - sandbox_bundles: Vec::new(), - cordoned: false, - cordon_owner: None, - cordon_reason: None, - retire_requested_at: None, - retired_at: None, - total_vcpus: 0, - wire_version: 0, - stages_images: false, - capabilities: engram_core::types::host::HostCapabilities::default(), - lease_expires_at: None, - lease_state: Default::default(), - lease_epoch: 0, - }); - meta.snapshots - .lock() - .push(engram_core::types::snapshot::SnapshotRecord { - id: engram_core::types::SnapshotId::new(), - session_id: Some(session_id), - host_id: Some(source_host), - image_version: "test".into(), - size_bytes: 1, - created_at: chrono::Utc::now(), - last_accessed_at: chrono::Utc::now(), - disk_manifest: None, - memory_manifest: Some(engram_core::types::manifest::ManifestRef::new()), - recoverable: true, - aux_bundles: Vec::new(), - events_cursor: None, - fc_snapshot_version: None, - }); - - let err = migrate_session_live(&state, session_id, target) - .await - .expect_err("dest failure must surface"); - assert!( - matches!(err, MigrateError::AbortedToSource(_)), - "got {err:?}", - ); - assert!( - abort_called.load(Ordering::SeqCst), - "source must be aborted" - ); - assert_eq!( - meta.get_session(session_id).await.unwrap().status, - SessionState::Active, - "session walks back to Active (zero loss)", - ); - } - - /// The commit-routing regression (prod canary 5fa742b7): step 4's - /// `invalidate_sandbox` + the PG rebind make the OLD sandbox id - /// unroutable, so a sandbox-routed `migration_commit` lands - /// "sandbox not found" and the frozen source lingers until the - /// export TTL. The verb must commit through the source backend - /// handle it resolved before freezing. - #[tokio::test] - async fn commit_reaches_the_frozen_source_after_rebind() { - use engram_core::types::snapshot::SnapshotMetadata; - use std::sync::atomic::{AtomicBool, Ordering}; - - struct MigratableHappyPath { - commit_called: Arc, - snapshot_called: Arc, - } - #[async_trait::async_trait] - impl engram_core::traits::SandboxBackend for MigratableHappyPath { - async fn create( - &self, - _: engram_core::types::sandbox::SandboxSpec, - ) -> Result { - Ok(SandboxId::new()) - } - async fn destroy(&self, _: SandboxId) -> Result<(), engram_core::SandboxError> { - Ok(()) - } - async fn list(&self) -> Result, engram_core::SandboxError> { - Ok(Vec::new()) - } - async fn exec_stream( - &self, - _: SandboxId, - _: engram_core::types::sandbox::ExecRequest, - ) -> Result - { - Err(engram_core::SandboxError::NotFound) - } - async fn snapshot( - &self, - _: SandboxId, - ) -> Result { - // Durability is the periodic cadence's job, not the - // finalize's — reaching here means the post-move Full - // regressed back in. - self.snapshot_called.store(true, Ordering::SeqCst); - Err(engram_core::SandboxError::NotFound) - } - async fn migration_presetup( - &self, - _: SandboxId, - ) -> Result - { - Ok(engram_core::types::snapshot::MigrationPresetupOut { - export_id: "test-export".into(), - peer_token: "test-token".into(), - peer_port: 9102, - sidecar_json: b"{}".to_vec(), - memory_manifest_json: b"{}".to_vec(), - memory_manifest_ref: engram_core::types::manifest::ManifestRef::new(), - disk_manifest_ref: None, - hot_chunks: vec![], - }) - } - async fn migration_capture_postcopy( - &self, - _: SandboxId, - _: &str, - ) -> Result - { - Ok(engram_core::types::snapshot::PostCopyCaptureOut { - sealed_chunks: 3, - total_chunks: 16, - pause_ms: 1, - disk_drain_ms: 1, - vmstate_ms: 1, - scan_ms: 1, - sealed_disk_chunks: 0, - paused_at_unix_ms: 0, - }) - } - async fn migration_drain_wait( - &self, - _: SandboxId, - ) -> Result - { - Ok(engram_core::types::snapshot::DrainOutcome::Done { - pulled: 3, - alt_sourced: 0, - zero_chunks: 0, - ms: 5, - }) - } - async fn migration_commit( - &self, - _: SandboxId, - _: &str, - ) -> Result<(), engram_core::SandboxError> { - self.commit_called.store(true, Ordering::SeqCst); - Ok(()) - } - async fn restore( - &self, - _: SnapshotMetadata, - ) -> Result { - Ok(SandboxId::new()) - } - fn snapshot_path_for(&self, _: engram_core::types::SnapshotId) -> std::path::PathBuf { - std::path::PathBuf::from("/nonexistent") - } - } - - let session = active_session(); - let session_id = session.id; - let source_host = session.host_id.expect("source host"); - let old_sandbox = session.sandbox_id.expect("source sandbox"); - let (state, meta, target) = build_state(session); - let commit_called = Arc::new(AtomicBool::new(false)); - let snapshot_called = Arc::new(AtomicBool::new(false)); - let happy: Arc = Arc::new(MigratableHappyPath { - commit_called: commit_called.clone(), - snapshot_called: snapshot_called.clone(), - }); - state.host_registry.register( - target, - Arc::new(engram_host_agent::host_client::LocalHostClient::with_noop_hub(happy.clone())), - ); - // The source resolves through the recorded sandbox owner (the - // same fake backend serves both roles, as in the abort test). - state - .host_registry - .record_sandbox_owner(old_sandbox, target); - meta.hosts - .lock() - .push(engram_core::types::host::HostRecord { - id: source_host, - hostname: "src".into(), - cloud_metadata: engram_core::types::host::HostMetadata::default(), - capacity: engram_core::types::host::HostCapacity { - total_gb: 100, - used_gb: 10, - total_mib: 65_536, - used_mib: 0, - running_sandboxes: 0, - }, - utilization: Default::default(), - status: engram_core::types::host::HostStatus::Ready, - last_heartbeat_at: chrono::Utc::now(), - host_addr: Some("http://127.0.0.1:1".into()), - ready_images: Vec::new(), - current_bundles: Vec::new(), - sandbox_bundles: Vec::new(), - cordoned: false, - cordon_owner: None, - cordon_reason: None, - retire_requested_at: None, - retired_at: None, - total_vcpus: 0, - wire_version: 0, - stages_images: false, - capabilities: engram_core::types::host::HostCapabilities::default(), - lease_expires_at: None, - lease_state: Default::default(), - lease_epoch: 0, - }); - meta.snapshots - .lock() - .push(engram_core::types::snapshot::SnapshotRecord { - id: engram_core::types::SnapshotId::new(), - session_id: Some(session_id), - host_id: Some(source_host), - image_version: "test".into(), - size_bytes: 1, - created_at: chrono::Utc::now(), - last_accessed_at: chrono::Utc::now(), - disk_manifest: None, - memory_manifest: Some(engram_core::types::manifest::ManifestRef::new()), - recoverable: true, - aux_bundles: Vec::new(), - events_cursor: None, - fc_snapshot_version: None, - }); - - migrate_session_live(&state, session_id, target) - .await - .expect("happy-path migration must succeed"); - // C2: the commit rides the FINALIZE task (after the drain) — - // poll for it instead of asserting synchronously. - let mut committed = false; - for _ in 0..60 { - if commit_called.load(Ordering::SeqCst) { - committed = true; - break; - } - tokio::time::sleep(std::time::Duration::from_millis(50)).await; - } - assert!( - committed, - "the frozen source must receive migration_commit (post-drain, \ - via the pre-resolved handle — the old sandbox id is \ - unroutable after the rebind)", - ); - let after = meta.get_session(session_id).await.unwrap(); - assert_eq!(after.host_id, Some(target), "session rebound to dest"); - assert_ne!( - after.sandbox_id, - Some(old_sandbox), - "session points at the new sandbox", - ); - // Durability is NOT migration's job (operator decision - // 2026-06-12): the finalize must NOT take a post-move - // checkpoint or record a row — the dest rides the periodic - // cadence, and until its first periodic capture, recovery - // rewinds to the source's last row (RPO ≤ cadence). Settle - // briefly so a regressed snapshot call (it followed commit in - // the same task) would have landed before the negative check. - tokio::time::sleep(std::time::Duration::from_millis(150)).await; - assert!( - !snapshot_called.load(Ordering::SeqCst), - "finalize took a post-move checkpoint — the guest-visible \ - post-move pause regressed back in", - ); - assert_eq!( - meta.snapshots.lock().len(), - 1, - "no migration-authored snapshot row; the seeded source row stays the recovery point", - ); - } - - /// A held lease refuses the migration outright (Fatal, not a - /// fallback — the rival owns the session right now). - /// R8: at most one in-flight migration per host endpoint; a second - /// claim sharing EITHER endpoint refuses, and Drop releases both - /// (including partial-claim rollback). - #[test] - fn migration_gate_claims_both_endpoints_and_releases_on_drop() { - let (a, b, c) = (HostId::new(), HostId::new(), HostId::new()); - let s1 = SessionId::new(); - let g = MigrationGateGuard::claim(Some(a), b, s1).expect("first claim"); - // Shares the source. - assert!(MigrationGateGuard::claim(Some(a), c, SessionId::new()).is_none()); - // Shares the dest. - assert!(MigrationGateGuard::claim(Some(c), b, SessionId::new()).is_none()); - // Disjoint hosts coexist. - let g2 = MigrationGateGuard::claim(None, c, SessionId::new()).expect("disjoint claim"); - drop(g); - // Released: both endpoints reusable; the partial-rollback path - // is exercised by the shares-the-dest refusal above (its `a` - // claim must have been rolled back). - let g3 = MigrationGateGuard::claim(Some(a), b, SessionId::new()).expect("after release"); - drop(g2); - drop(g3); - // (No global-emptiness assert: the gate is a process-global and - // sibling tests claim it concurrently; release is proven by the - // successful re-claim above.) - } - - /// Regression guard for the rollback-path shard deadlock: the - /// `Occupied` arm used to `remove` rolled-back claims while still - /// holding the current host's `entry()` shard lock, which hangs - /// when two hosts collide on a shard. Run the rollback sequence - /// enough times with fresh random ids that a same-shard collision - /// is near-certain — a reintroduced deadlock hangs the whole suite. - #[test] - fn claim_rollback_never_deadlocks_on_shard_collision() { - for _ in 0..5000 { - let dest = HostId::new(); - let held = MigrationGateGuard::claim(None, dest, SessionId::new()).expect("hold dest"); - // `claim(Some(src), dest, …)` claims src (vacant) then hits - // dest (occupied) → rolls back src while the dest entry - // guard is live. If src and dest share a shard, the old - // code deadlocked here. - let src = HostId::new(); - assert!( - MigrationGateGuard::claim(Some(src), dest, SessionId::new()).is_none(), - "dest is held; the claim must fail and roll back src cleanly", - ); - // src must be fully released by the rollback — re-claimable. - let reclaim = MigrationGateGuard::claim(Some(src), HostId::new(), SessionId::new()) - .expect("rolled-back src is reusable"); - drop(reclaim); - drop(held); - } - } - - /// Issue #209 (CASE 2 + CASE 1): a successfully-migrated session must - /// NOT be abandoned Active-on-the-dest when the finalize's - /// `migration_drain_wait` keeps returning a transport `Err`. The - /// pre-fix `Err` arm just logged "leaving the source to its TTL" and - /// `return`ed — so the source's export TTL fired, the ownership probe - /// answered owned=false, and the host destroyed the page server the - /// dest still needed, wedging the session Active on a poisoned VM. - /// - /// With the fix the drain is retried with bounded backoff; on - /// persistent failure the finalize takes the SAME rung-1 rewind as - /// the PeerLost arm: the dest sandbox is destroyed and the session is - /// re-armed Evacuating for the scanner to rehome from the durable - /// row. It also asserts (CASE 1) the teleport_target pin is cleared — - /// it must never leak past a verb exit, since it strictly pins every - /// future rehome of the session to the chosen host. - #[tokio::test(flavor = "multi_thread", worker_threads = 2)] - async fn drain_transport_error_rewinds_instead_of_abandoning_active() { - use engram_core::types::snapshot::SnapshotMetadata; - use std::sync::atomic::{AtomicBool, AtomicU32, Ordering}; - - std::env::set_var("ENGRAM_LIVE_TELEPORT", "1"); - // Collapse the retry backoff so the exhaustion path is fast. - std::env::set_var("ENGRAM_MIGRATION_DRAIN_RETRY_BASE_MS", "1"); - - struct DrainFlakyDest { - destroy_called: Arc, - commit_called: Arc, - drain_attempts: Arc, - } - #[async_trait::async_trait] - impl engram_core::traits::SandboxBackend for DrainFlakyDest { - async fn create( - &self, - _: engram_core::types::sandbox::SandboxSpec, - ) -> Result { - Ok(SandboxId::new()) - } - async fn destroy(&self, _: SandboxId) -> Result<(), engram_core::SandboxError> { - self.destroy_called.store(true, Ordering::SeqCst); - Ok(()) - } - async fn list(&self) -> Result, engram_core::SandboxError> { - Ok(Vec::new()) - } - async fn exec_stream( - &self, - _: SandboxId, - _: engram_core::types::sandbox::ExecRequest, - ) -> Result - { - Err(engram_core::SandboxError::NotFound) - } - async fn snapshot( - &self, - _: SandboxId, - ) -> Result { - Err(engram_core::SandboxError::NotFound) - } - async fn migration_presetup( - &self, - _: SandboxId, - ) -> Result - { - Ok(engram_core::types::snapshot::MigrationPresetupOut { - export_id: "test-export".into(), - peer_token: "test-token".into(), - peer_port: 9102, - sidecar_json: b"{}".to_vec(), - memory_manifest_json: b"{}".to_vec(), - memory_manifest_ref: engram_core::types::manifest::ManifestRef::new(), - disk_manifest_ref: None, - hot_chunks: vec![], - }) - } - async fn migration_capture_postcopy( - &self, - _: SandboxId, - _: &str, - ) -> Result - { - Ok(engram_core::types::snapshot::PostCopyCaptureOut { - sealed_chunks: 3, - total_chunks: 16, - pause_ms: 1, - disk_drain_ms: 1, - vmstate_ms: 1, - scan_ms: 1, - sealed_disk_chunks: 0, - paused_at_unix_ms: 0, - }) - } - async fn migration_drain_wait( - &self, - _: SandboxId, - ) -> Result - { - // ALWAYS a transport error — never PeerLost, never Done. - // The fix must retry, then (budget exhausted) rewind. - self.drain_attempts.fetch_add(1, Ordering::SeqCst); - Err(engram_core::SandboxError::Unavailable( - "injected transport blip on drain_wait".into(), - )) - } - async fn migration_commit( - &self, - _: SandboxId, - _: &str, - ) -> Result<(), engram_core::SandboxError> { - // The source must NOT be committed (destroyed) when the - // drain never completed — the rewind owns recovery. - self.commit_called.store(true, Ordering::SeqCst); - Ok(()) - } - async fn restore( - &self, - _: SnapshotMetadata, - ) -> Result { - Ok(SandboxId::new()) - } - fn snapshot_path_for(&self, _: engram_core::types::SnapshotId) -> std::path::PathBuf { - std::path::PathBuf::from("/nonexistent") - } - } - - let session = active_session(); - let session_id = session.id; - let source_host = session.host_id.expect("source host"); - let old_sandbox = session.sandbox_id.expect("source sandbox"); - let (state, meta, target) = build_state(session); - let destroy_called = Arc::new(AtomicBool::new(false)); - let commit_called = Arc::new(AtomicBool::new(false)); - let drain_attempts = Arc::new(AtomicU32::new(0)); - let flaky: Arc = Arc::new(DrainFlakyDest { - destroy_called: destroy_called.clone(), - commit_called: commit_called.clone(), - drain_attempts: drain_attempts.clone(), - }); - state.host_registry.register( - target, - Arc::new(engram_host_agent::host_client::LocalHostClient::with_noop_hub(flaky.clone())), - ); - state - .host_registry - .record_sandbox_owner(old_sandbox, target); - meta.hosts - .lock() - .push(engram_core::types::host::HostRecord { - id: source_host, - hostname: "src".into(), - cloud_metadata: engram_core::types::host::HostMetadata::default(), - capacity: engram_core::types::host::HostCapacity { - total_gb: 100, - used_gb: 10, - total_mib: 65_536, - used_mib: 0, - running_sandboxes: 0, - }, - utilization: Default::default(), - status: engram_core::types::host::HostStatus::Ready, - last_heartbeat_at: chrono::Utc::now(), - host_addr: Some("http://127.0.0.1:1".into()), - ready_images: Vec::new(), - current_bundles: Vec::new(), - sandbox_bundles: Vec::new(), - cordoned: false, - cordon_owner: None, - cordon_reason: None, - retire_requested_at: None, - retired_at: None, - total_vcpus: 0, - wire_version: 0, - stages_images: false, - capabilities: engram_core::types::host::HostCapabilities::default(), - lease_expires_at: None, - lease_state: Default::default(), - lease_epoch: 0, - }); - meta.snapshots - .lock() - .push(engram_core::types::snapshot::SnapshotRecord { - id: engram_core::types::SnapshotId::new(), - session_id: Some(session_id), - host_id: Some(source_host), - image_version: "test".into(), - size_bytes: 1, - created_at: chrono::Utc::now(), - last_accessed_at: chrono::Utc::now(), - disk_manifest: None, - memory_manifest: Some(engram_core::types::manifest::ManifestRef::new()), - recoverable: true, - aux_bundles: Vec::new(), - events_cursor: None, - fc_snapshot_version: None, - }); - - // The verb returns Ok — the move LANDED; the finalize runs async. - migrate_session_live(&state, session_id, target) - .await - .expect("the move lands; the drain finalize runs in the background"); - - // The finalize retries the drain, exhausts the budget, and takes - // the rung-1 rewind: the dest sandbox is destroyed and the - // session is re-armed Evacuating (durable row present → parachute, - // not kill). Poll for the terminal posture. - let mut rewound = false; - for _ in 0..400 { - if destroy_called.load(Ordering::SeqCst) - && meta.get_session(session_id).await.unwrap().status == SessionState::Evacuating - { - rewound = true; - break; - } - tokio::time::sleep(std::time::Duration::from_millis(25)).await; - } - assert!( - rewound, - "issue #209: a persistent drain transport error must rewind \ - (destroy dest + re-arm Evacuating), never abandon the session \ - Active on the dest while the source page server is reaped", - ); - assert!( - drain_attempts.load(Ordering::SeqCst) > 1, - "the drain must be RETRIED before giving up, not abandoned on \ - the first transport error (saw {} attempt(s))", - drain_attempts.load(Ordering::SeqCst), - ); - assert!( - !commit_called.load(Ordering::SeqCst), - "the source must NOT be committed/destroyed when the drain never \ - completed — the rewind owns recovery from the durable row", - ); - // CASE 1: the teleport_target pin must be cleared — it must never - // outlive the verb (it strictly pins every future rehome). - assert_eq!( - meta.get_teleport_target(session_id).await.unwrap(), - None, - "issue #209 CASE 1: the teleport_target pin leaked past the verb", - ); - } - - #[tokio::test] - async fn busy_op_lane_refuses_migration() { - let session = active_session(); - let session_id = session.id; - let (state, meta, target) = build_state(session); - meta.ops - .seed_running(session_id, engram_core::types::session_op::OpKind::Evict); - let err = migrate_session_live(&state, session_id, target) - .await - .expect_err("a running op must refuse the migration"); - assert!(matches!(err, MigrateError::Fatal(_))); - } -} diff --git a/crates/engram-coordinator/src/placement.rs b/crates/engram-coordinator/src/placement.rs index bfc3a8aeb..9cbcf7303 100644 --- a/crates/engram-coordinator/src/placement.rs +++ b/crates/engram-coordinator/src/placement.rs @@ -973,41 +973,6 @@ pub async fn pick_for_session( result } -/// #800: the RESERVED evac/resume placement — [`pick_for_session`] with the -/// capacity-soft fallback dropped. Returns `NoCapacity` when no schedulable -/// host fits the session's 2D budget (rather than binding the first-ranked, -/// possibly measured-full host), so the caller can QUEUE the session via the -/// reserved queue path (the #795 resume precedent) and re-home it once -/// capacity returns. The one commit point that mutates reservations stays -/// the caller's rebind (`assign_session_host`); this makes the *decision* -/// honor the hard bound. Emits the same K4 demand-pressure counter. -pub async fn pick_for_session_reserved( - meta: &dyn MetadataStore, - registry: &HostRegistry, - ctx: &ScheduleContext<'_>, - now: DateTime, -) -> Result<(HostId, Arc), PickError> { - let result = pick_for_session_inner(meta, registry, ctx, now, true).await; - let outcome = match &result { - Ok(_) => "placed", - Err(PickError::NoCapacity) => "no_capacity", - Err(PickError::ImageNotReady(_)) => "image_not_ready", - Err(PickError::HostUnreachable(..)) => "host_unreachable", - Err(PickError::Internal(_)) => "internal", - }; - ::metrics::counter!(crate::metrics::SESSION_PLACEMENT_TOTAL, "outcome" => outcome).increment(1); - // On NoCapacity, name the per-host exclusion reasons (present-but-full - // hosts show up via the ranked set; a fully-cordoned fleet via the - // empty-candidate summary) — the reserved NoCapacity is the QUEUE - // signal, not a mystery stall, but the visibility is still useful. - if matches!(result, Err(PickError::NoCapacity)) { - // `origin="resume"` — the bounded PLACEMENT_EXCLUDED_TOTAL vocabulary - // already scopes the resume/evac path under this label (metrics.rs). - log_empty_candidates(meta, ctx, "resume", now).await; - } - result -} - async fn pick_for_session_inner( meta: &dyn MetadataStore, registry: &HostRegistry, diff --git a/crates/engram-coordinator/src/queue_scanner.rs b/crates/engram-coordinator/src/queue_scanner.rs index c9d6c063d..0645d9358 100644 --- a/crates/engram-coordinator/src/queue_scanner.rs +++ b/crates/engram-coordinator/src/queue_scanner.rs @@ -87,7 +87,7 @@ //! `spawn` also parks on the same `wake`-or-`poll_interval` race before //! its very first sweep (not just between sweeps) — a beat for hosts to //! heartbeat back in on a cold coordinator start, same rationale as -//! `evac_resumer::spawn`'s skip-the-first-tick. +//! `teleport::spawn`'s skip-the-first-tick. use std::collections::HashMap; use std::sync::Arc; @@ -186,7 +186,7 @@ fn partition_queue(queued: Vec) -> Vec> { classes } -/// Spawn the queue scanner. Mirrors [`crate::evac_resumer::spawn`], plus +/// Spawn the queue scanner. Mirrors [`crate::teleport::spawn`], plus /// the shared `wake` handle: `pg_listener` fires it on `placement_changed` /// NOTIFYs so a sweep runs as soon as capacity might have freed, instead /// of waiting for `poll_interval`. @@ -196,7 +196,7 @@ pub fn spawn( wake: Arc, ) -> tokio::task::JoinHandle<()> { tokio::spawn(async move { - // Mirrors `evac_resumer::spawn`'s skip-the-first-immediate-tick: + // Mirrors `teleport::spawn`'s skip-the-first-immediate-tick: // coord just started, so don't sweep at the very first instant. // `wake` still lets a real `placement_changed` NOTIFY (a host // registering, a session enqueuing — all written straight to the diff --git a/crates/engram-coordinator/src/session_ops.rs b/crates/engram-coordinator/src/session_ops.rs index ea785caa6..3074d90a5 100644 --- a/crates/engram-coordinator/src/session_ops.rs +++ b/crates/engram-coordinator/src/session_ops.rs @@ -388,24 +388,6 @@ impl OpClaim { } } - /// Stamp progress (heartbeat) on the running row. The tri-state - /// mirrors the retired lease's `touch_checked`: `Held` = still ours, - /// `Lost` = a successor re-claimed (authoritative — stop), - /// `TransientError` = a PG blip the caller may retry through. - pub(crate) async fn touch(&self, step: &str) -> OpTouch { - match self - .state - .services - .meta - .op_record_step(self.op.id, self.epoch, step) - .await - { - Ok(true) => OpTouch::Held, - Ok(false) => OpTouch::Lost, - Err(e) => OpTouch::TransientError(e), - } - } - /// Background heartbeat for straight-line pipelines with no touch /// loop of their own (the manual snapshot's capture body). Keeps the /// running row's `heartbeat_at` fresh so the reclaim sweep (180s @@ -544,15 +526,6 @@ impl Drop for OpHeartbeat { } } -/// Outcome of an [`OpClaim::touch`]. `Lost` is authoritative (a -/// successor re-claimed the op — give up ownership); `TransientError` is -/// a PG-transport blip the caller may retry through. -pub(crate) enum OpTouch { - Held, - Lost, - TransientError(engram_core::MetaError), -} - /// The executor loop: one per coordinator pod. Wakes on /// `pg_notify('session_ops', …)` (via `wake`) or the fallback tick, /// claims head ops per due session, and drives them. Also owns the diff --git a/crates/engram-coordinator/src/session_verbs.rs b/crates/engram-coordinator/src/session_verbs.rs index 228ea9856..2d98de5a8 100644 --- a/crates/engram-coordinator/src/session_verbs.rs +++ b/crates/engram-coordinator/src/session_verbs.rs @@ -27,16 +27,13 @@ pub async fn dispatch(ctx: &OpCtx<'_>) -> OpOutcome { // in follow-up phases), so terminally fail the row — this FREES // the session's op lane (fence-then-free, the successor to the // retired lease reaper) and the existing recovery machinery owns - // the rest (periodic checkpoints for a torn manual snapshot; the - // ADR 0018 parachute / evac scanner for a torn teleport). + // the rest (periodic checkpoints for a torn manual snapshot). A + // Teleport op is re-drivable: `teleport::drive` resumes from the + // `session_teleports` row (ADR 0123 B). OpKind::CheckpointFinalize => OpOutcome::Failed( "manual snapshot claim abandoned (holder died); safe to retry the snapshot".into(), ), - OpKind::Teleport => OpOutcome::Failed( - "teleport claim abandoned (holder died); the parachute/evac machinery owns \ - recovery — safe to re-issue the move" - .into(), - ), + OpKind::Teleport => crate::teleport::drive(ctx).await, } } @@ -146,7 +143,7 @@ fn derive_resume_plan( /// /// `require_confirm` selects the teardown posture: the deterministic /// spawn re-plan runs against a LIVE host, so the unbind is gated on -/// `confirm_source_teardown` (host-affirmed release — never strand a +/// `destroy_retained_sandbox` (host-affirmed release — never strand a /// running VM's plane behind a cleared row); the Unreachable arm's /// guest is already dead, so its destroy stays best-effort and the /// tombstone alone owns cleanup. @@ -171,7 +168,7 @@ async fn rebuild_binding( return OpOutcome::Retry("rebuild: tombstone write failed".into()); } if require_confirm { - if let Err(e) = crate::evac_resumer::confirm_source_teardown( + if let Err(e) = crate::api::snapshot::destroy_retained_sandbox( state, host_id, sandbox_id, diff --git a/crates/engram-coordinator/src/state.rs b/crates/engram-coordinator/src/state.rs index 90119fa8d..b047d02a1 100644 --- a/crates/engram-coordinator/src/state.rs +++ b/crates/engram-coordinator/src/state.rs @@ -28,6 +28,14 @@ use crate::Services; #[derive(Clone, Debug, Deserialize, Serialize)] #[serde(tag = "type", rename_all = "snake_case")] pub enum SessionEvent { + TeleportFinished { + teleport_id: engram_core::TeleportId, + outcome: engram_core::types::teleport::TeleportOutcome, + kind: engram_core::types::teleport::TeleportKind, + dest_host_id: Option, + error: Option, + at: DateTime, + }, /// Session moved between lifecycle states. StatusChanged { from: SessionState, @@ -504,6 +512,7 @@ impl SessionEvent { pub fn kind(&self) -> &'static str { match self { Self::StatusChanged { .. } => "status_changed", + Self::TeleportFinished { .. } => "teleport_finished", Self::PromptReceived { .. } => "prompt_received", Self::ToolResultSubmitted { .. } => "tool_result_submitted", Self::HarnessModeChanged { .. } => "harness_mode_changed", @@ -2281,24 +2290,8 @@ pub(crate) mod tests { /// `assign_session_sandbox(None)`) so tests can verify the /// barrier behaviour Phase C will rely on. pub(crate) chunk_generation: PlMutex, - /// ADR 0018 commit 12b: in-memory mirror of - /// `sessions.evac_attempts`. Reset to 0 when the session - /// transitions into Evacuating; bumped by - /// `bump_evac_attempts`; observed by the scanner via - /// `list_evacuating_sessions`. - pub(crate) evac_attempts: PlMutex>, - /// ADR 0034: in-memory mirror of `sessions.evict_attempts`. - /// Same lifecycle as `evac_attempts`, for the eviction - /// scanner. + /// ADR 0034: in-memory mirror of the eviction retry count. pub(crate) evict_attempts: PlMutex>, - /// ADR 0047: in-memory mirror of `teleport_targets` (the - /// migration pin honored by evac_resumer as a strict - /// require_host). Issue #209 tests assert this is cleared on - /// every verb exit path — the default no-op trait impl would make - /// such an assertion vacuous, so the mock tracks it for real. - /// Issue #214: tracks the set-at timestamp alongside the pin so - /// the aged-pin scanner test can backdate one deterministically. - pub(crate) teleport_targets: PlMutex, /// Issue #531/PR #564 (ADR 0068 persist-before-reconcile /// regression): when true, the NEXT `touch_host_heartbeat` call /// fails instead of persisting — tests use this to prove the @@ -2340,15 +2333,6 @@ pub(crate) mod tests { pub(crate) harness: PlMutex>, } - /// Alias so `clippy::type_complexity` stays happy on MiniMeta's - /// `teleport_targets` field. `session_id → (target_host, set_at?)` — - /// `set_at` is `None` only for a pin staged before issue #214's - /// migration (the scanner treats such a pin as not-aged). - pub(crate) type TeleportPinMap = std::collections::HashMap< - SessionId, - (engram_core::HostId, Option>), - >; - impl MiniMeta { /// ADR 0047: placement reads host rows now — tests stage a /// schedulable (ready, fresh-heartbeat) host with this. @@ -2397,9 +2381,7 @@ pub(crate) mod tests { fail_next_record_snapshot: PlMutex::new(false), live_disk_manifests: PlMutex::new(std::collections::HashMap::new()), chunk_generation: PlMutex::new(0), - evac_attempts: PlMutex::new(std::collections::HashMap::new()), evict_attempts: PlMutex::new(std::collections::HashMap::new()), - teleport_targets: PlMutex::new(std::collections::HashMap::new()), fail_next_heartbeat_persist: PlMutex::new(false), fail_next_append_event: PlMutex::new(false), reconcile_probe_calls: PlMutex::new(0), @@ -2589,9 +2571,6 @@ pub(crate) mod tests { // budget clean. Mirrors the PG `CASE WHEN $2 = // 'evacuating' THEN 0` branch in // `engram_postgres::transition_session`. - if matches!(target, engram_core::types::SessionState::Evacuating) { - self.evac_attempts.lock().insert(id, 0); - } // ADR 0034: same reset-on-entry for Evicting. Mirrors the // PG `CASE WHEN $2 = 'evicting' THEN 0` branch. if matches!(target, engram_core::types::SessionState::Evicting) { @@ -2665,28 +2644,7 @@ pub(crate) mod tests { s.host_id = host_id; Ok(()) } - async fn set_teleport_target( - &self, - id: engram_core::SessionId, - target: Option, - ) -> Result<(), MetaError> { - let mut t = self.teleport_targets.lock(); - match target { - Some(h) => { - t.insert(id, (h, Some(chrono::Utc::now()))); - } - None => { - t.remove(&id); - } - } - Ok(()) - } - async fn get_teleport_target( - &self, - id: engram_core::SessionId, - ) -> Result>)>, MetaError> { - Ok(self.teleport_targets.lock().get(&id).copied()) - } + async fn assign_session_sandbox( &self, id: engram_core::SessionId, @@ -2780,6 +2738,38 @@ pub(crate) mod tests { _ => Vec::new(), }) } + async fn get_host(&self, id: HostId) -> Result, MetaError> { + Ok(self.hosts.lock().iter().find(|h| h.id == id).cloned()) + } + async fn open_teleport_for_session( + &self, + _id: SessionId, + ) -> Result, MetaError> { + Ok(None) + } + async fn request_host_retirement( + &self, + id: HostId, + owner: engram_core::types::host::CordonOwner, + reason: &str, + now: chrono::DateTime, + ) -> Result { + use engram_core::types::host::HostStatus; + let mut hosts = self.hosts.lock(); + let Some(h) = hosts.iter_mut().find(|h| h.id == id) else { + return Ok(false); + }; + if !matches!(h.status, HostStatus::Ready | HostStatus::Draining) + || h.cordon_owner.is_some_and(|o| o != owner) + { + return Ok(false); + } + h.cordoned = true; + h.cordon_owner = Some(owner); + h.cordon_reason = Some(reason.to_owned()); + h.retire_requested_at.get_or_insert(now); + Ok(true) + } async fn set_host_cordon( &self, id: HostId, @@ -3474,25 +3464,6 @@ pub(crate) mod tests { // ADR 0018 commit 12b: scanner support. MiniMeta carries one // session, so the list-sweep is trivially "is it Evacuating?". - async fn list_evacuating_sessions(&self) -> Result, MetaError> { - let s = self.session.lock().clone(); - if matches!(s.status, engram_core::types::SessionState::Evacuating) { - let attempts = self.evac_attempts.lock().get(&s.id).copied().unwrap_or(0); - Ok(vec![(s, attempts)]) - } else { - Ok(Vec::new()) - } - } - - async fn bump_evac_attempts( - &self, - session_id: engram_core::SessionId, - ) -> Result { - let mut map = self.evac_attempts.lock(); - let entry = map.entry(session_id).or_insert(0); - *entry += 1; - Ok(*entry) - } // ADR 0034: eviction-scanner support, mirroring the 12b evac // trio above. MiniMeta carries one session, so the list-sweep diff --git a/crates/engram-coordinator/src/teleport.rs b/crates/engram-coordinator/src/teleport.rs new file mode 100644 index 000000000..30ebbdd4c --- /dev/null +++ b/crates/engram-coordinator/src/teleport.rs @@ -0,0 +1,1352 @@ +//! Durable teleport driver (ADR 0123 B). +//! +//! Teleport payloads are either `{source_host, reason}` for planned admission, +//! or `{teleport_id}` for a move that has already been admitted. The row's +//! phase is the recovery cursor. An op step marker never replaces that row. + +use crate::error::ApiError; +use crate::session_ops::{OpCtx, OpOutcome}; +use crate::state::{SessionEvent, SharedState}; +use engram_core::types::host::{HostStatus, RetirementGrant}; +use engram_core::types::session_op::{OpKind, OpState}; +use engram_core::types::snapshot::{SnapshotMetadata, SnapshotRecord}; +use engram_core::types::teleport::TeleportSettle; +use engram_core::types::teleport::*; +use engram_core::types::{BindingDisposition, SessionState}; +use engram_core::{HostId, SandboxError, SessionId}; +use std::time::Duration; + +#[derive(Clone, Debug)] +pub struct TeleportConfig { + pub poll_interval: Duration, + pub max_open_per_dest: u32, +} +impl Default for TeleportConfig { + fn default() -> Self { + Self { + poll_interval: Duration::from_secs(10), + max_open_per_dest: 1, + } + } +} +pub fn spawn(cfg: TeleportConfig, state: SharedState) -> tokio::task::JoinHandle<()> { + tokio::spawn(async move { + let mut timer = tokio::time::interval(cfg.poll_interval); + timer.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + loop { + timer.tick().await; + if let Err(e) = run_once(&cfg, &state).await { + tracing::warn!(error=%e,"teleport scan failed"); + } + } + }) +} +pub async fn run_once( + cfg: &TeleportConfig, + state: &SharedState, +) -> Result<(), Box> { + for host in state.services.meta.list_retiring_hosts().await? { + plan_host_teleports_with_limit( + state, + host.id, + TeleportReason::RetireHost, + cfg.max_open_per_dest, + ) + .await; + if matches!( + state + .services + .meta + .grant_host_retirement(host.id, state.services.clock.now_utc()) + .await?, + RetirementGrant::Granted + ) { + tracing::info!(host_id=%host.id,"host retirement granted"); + metrics::counter!("engram_host_retirement_granted_total").increment(1); + } + } + for row in state.services.meta.list_open_teleports().await? { + if !state + .services + .meta + .op_pending_exists(row.session_id, OpKind::Teleport) + .await? + { + enqueue_deferred( + state, + row.session_id, + OpKind::Teleport, + serde_json::json!({"teleport_id":row.id,"max_open_per_dest":cfg.max_open_per_dest}), + Some(&format!("teleport:{}", row.id)), + ) + .await?; + } + } + Ok(()) +} +#[derive(Default)] +pub(crate) struct PlanReport { + pub planned: Vec, + pub descended: Vec, + pub skipped: u32, +} +pub(crate) async fn plan_host_teleports( + state: &SharedState, + host: HostId, + reason: TeleportReason, +) -> PlanReport { + plan_host_teleports_with_limit( + state, + host, + reason, + TeleportConfig::default().max_open_per_dest, + ) + .await +} +async fn plan_host_teleports_with_limit( + state: &SharedState, + host: HostId, + reason: TeleportReason, + max_open_per_dest: u32, +) -> PlanReport { + let mut report = PlanReport::default(); + let assignments = match state + .services + .meta + .list_resident_sandbox_assignments_on_host(host) + .await + { + Ok(rows) => rows, + Err(e) => { + tracing::warn!(%host,error=%e,"teleport planning failed"); + return report; + } + }; + for (id, _, status) in assignments { + let result = match status { + SessionState::Active => enqueue_deferred( + state, + id, + OpKind::Teleport, + serde_json::json!({"source_host":host,"reason":reason,"max_open_per_dest":max_open_per_dest}), + Some(&format!("teleport:{host}:{id}")), + ) + .await + .map(|_| { + report.planned.push(id); + }), + SessionState::Parked => match state.services.meta.get_session(id).await { + Ok(s) => crate::idle_evictor::descend_parked_session(state, &s, "teleport") + .await + .map(|done| { + if done { + report.descended.push(id) + } else { + report.skipped += 1 + } + }), + Err(e) => Err(e), + }, + _ => { + report.skipped += 1; + continue; + } + }; + if let Err(e) = result { + report.skipped += 1; + tracing::warn!(session_id=%id,error=%e,"teleport planning failed"); + } + } + report +} + +async fn enqueue_deferred( + state: &SharedState, + id: SessionId, + kind: OpKind, + payload: serde_json::Value, + key: Option<&str>, +) -> Result<(), engram_core::MetaError> { + if let engram_core::types::session_op::EnqueueOutcome::Claimed(op) = + crate::session_ops::enqueue_claim(state, id, kind, payload, key).await? + { + state + .services + .meta + .op_requeue_with_backoff( + op.id, + op.epoch.expect("claimed epoch"), + Duration::ZERO, + "teleport scheduled", + ) + .await?; + } + Ok(()) +} + +pub(crate) async fn admit_for_rpc( + state: &SharedState, + id: SessionId, + target: Option, +) -> Result { + let claim=crate::session_ops::OpClaim::try_acquire(state,id,OpKind::Teleport,serde_json::json!({"source_host":state.services.meta.get_session(id).await?.host_id,"reason":TeleportReason::Ui})).await? + .ok_or_else(||ApiError::Conflict("busy_lane".into()))?; + let admitted = steps::admit(&claim.as_ctx(), target, TeleportReason::Ui).await; + match admitted { + Ok(TeleportAdmitOutcome::Admitted(row)) => { + // Queue the durable row while the exclusive admission claim still owns the lane. + enqueue_deferred( + state, + id, + OpKind::Teleport, + serde_json::json!({"teleport_id":row.id}), + Some(&format!("teleport:{}", row.id)), + ) + .await?; + claim.finish(OpState::Done, None).await; + Ok(*row) + } + other => { + claim.finish(OpState::Done, None).await; + match other { + Ok(TeleportAdmitOutcome::NoFit) => Err(ApiError::Conflict("no_fit".into())), + Ok(TeleportAdmitOutcome::SessionNotActive(_)) => { + Err(ApiError::Conflict("not_active".into())) + } + Ok(TeleportAdmitOutcome::Fenced) => Err(ApiError::Conflict("busy_lane".into())), + Err(e) => Err(e), + Ok(TeleportAdmitOutcome::Admitted(_)) => unreachable!(), + } + } + } +} + +pub(crate) async fn drive(ctx: &OpCtx<'_>) -> OpOutcome { + match drive_inner(ctx).await { + Ok(out) => out, + Err(e) => OpOutcome::Retry(e.to_string()), + } +} +async fn drive_inner(ctx: &OpCtx<'_>) -> Result { + loop { + let Some(row) = ctx + .state + .services + .meta + .open_teleport_for_session(ctx.op.session_id) + .await? + else { + if ctx.op.payload.get("teleport_id").is_some() { + return Ok(OpOutcome::Done); + } + let s = ctx + .state + .services + .meta + .get_session(ctx.op.session_id) + .await?; + let source: Option = ctx + .op + .payload + .get("source_host") + .and_then(|v| serde_json::from_value(v.clone()).ok()); + if source.is_none() || source != s.host_id { + return Ok(OpOutcome::Done); + } + let reason = ctx + .op + .payload + .get("reason") + .and_then(|v| serde_json::from_value(v.clone()).ok()) + .unwrap_or(TeleportReason::RetireHost); + if !matches!( + steps::admit(ctx, None, reason).await?, + TeleportAdmitOutcome::Admitted(_) + ) { + return Ok(OpOutcome::Done); + } + continue; + }; + if !ctx.step(row.phase.as_str()).await { + return Ok(OpOutcome::Done); + } + let session = ctx.state.services.meta.get_session(row.session_id).await?; + // A crash can occur between the fenced settle and the terminal row CAS. + if session.sandbox_id.is_none() + && matches!(session.status, SessionState::Idle | SessionState::Dead) + { + let error = match row.phase { + TeleportPhase::RollingBack => "source_lost_during_rollback", + TeleportPhase::Attached => "peer_lost", + _ => "dest_lost_after_blackout", + }; + ctx.state + .services + .meta + .teleport_settle( + row.id, + ctx.epoch, + TeleportSettle { + error: error.into(), + session: None, + entomb_source: false, + }, + ) + .await?; + return Ok(OpOutcome::Done); + } + let outcome = match row.phase { + TeleportPhase::Admitted => steps::capture(ctx, &row).await?, + TeleportPhase::Captured => steps::restore(ctx, &row).await?, + TeleportPhase::Restored => steps::commit(ctx, &row).await?, + TeleportPhase::Committed => steps::attach(ctx, &row).await?, + TeleportPhase::Attached => steps::release(ctx, &row).await?, + TeleportPhase::RollingBack => steps::rollback(ctx, &row).await?, + TeleportPhase::Done | TeleportPhase::Aborted | TeleportPhase::Failed => { + return Ok(OpOutcome::Done) + } + }; + if let Some(outcome) = outcome { + return Ok(outcome); + } + } +} +async fn advance( + ctx: &OpCtx<'_>, + row: &TeleportRow, + to: TeleportPhase, + patch: TeleportPatch, +) -> Result, ApiError> { + Ok((!ctx + .state + .services + .meta + .teleport_advance(row.id, row.phase, to, patch, ctx.epoch) + .await?) + .then_some(OpOutcome::Done)) +} +async fn rollback_begin( + ctx: &OpCtx<'_>, + row: &TeleportRow, + error: String, +) -> Result, ApiError> { + advance( + ctx, + row, + TeleportPhase::RollingBack, + TeleportPatch { + error: Some(error), + ..Default::default() + }, + ) + .await +} +/// A dead or retired host, or one whose row was deleted: nothing can +/// answer for the sandbox any more. +async fn source_gone(ctx: &OpCtx<'_>, row: &TeleportRow) -> Result { + Ok(ctx + .state + .services + .meta + .get_host(row.source_host_id) + .await? + .is_none_or(|h| matches!(h.status, HostStatus::Dead | HostStatus::Retired))) +} + +mod steps { + use super::*; + async fn live_metadata( + _ctx: &OpCtx<'_>, + row: &TeleportRow, + ) -> Result { + serde_json::from_value( + row.live_payload + .clone() + .ok_or_else(|| ApiError::Internal("live teleport has no restore payload".into()))?, + ) + .map_err(|e| ApiError::Internal(format!("live restore payload: {e}"))) + } + async fn capture_live( + ctx: &OpCtx<'_>, + row: &TeleportRow, + ) -> Result, ApiError> { + let state = ctx.state; + let host = match state.host_registry.backend_for(row.source_host_id).await { + Ok(host) => host, + Err(e) => return rollback_begin(ctx, row, e.to_string()).await, + }; + if row.export_id.is_none() { + let presetup = match host + .migration_presetup(row.source_sandbox_id, ctx.fence()) + .await + { + Ok(p) => p, + Err(SandboxError::InvalidSpec(_)) => { + return advance( + ctx, + row, + TeleportPhase::Admitted, + TeleportPatch { + kind: Some(TeleportKind::Snapshot), + ..Default::default() + }, + ) + .await + } + Err(e) => return rollback_begin(ctx, row, e.to_string()).await, + }; + let session = state.services.meta.get_session(row.session_id).await?; + let durable = state + .services + .meta + .latest_snapshot_for_session(row.session_id) + .await?; + let source_addr = state + .services + .meta + .get_host(row.source_host_id) + .await? + .and_then(|h| h.host_addr) + .ok_or_else(|| ApiError::Unavailable("source host has no address".into()))?; + let hostpart = source_addr + .trim_start_matches("http://") + .trim_start_matches("https://"); + let hostname = hostpart.rsplit_once(':').map_or(hostpart, |(h, _)| h); + let peer_addr = format!("{hostname}:{}", presetup.peer_port); + let metadata = SnapshotMetadata { + id: engram_core::SnapshotId::from(state.services.entropy.uuid()), + size_bytes: durable.as_ref().map_or(0, |s| s.size_bytes), + created_at: state.services.clock.now_utc(), + image_version: durable + .as_ref() + .map_or_else(|| session.image.clone(), |s| s.image_version.clone()), + disk_manifest: presetup.disk_manifest_ref, + memory_manifest: Some(presetup.memory_manifest_ref), + base_memory_manifest: crate::api::snapshot::base_memory_manifest_for_image( + state, + &session.image, + ) + .await, + migration_source: Some(engram_core::types::snapshot::MigrationSourceInfo { + export_id: presetup.export_id.clone(), + source_addr, + memory_manifest_json: presetup.memory_manifest_json, + disk_manifest_json: Vec::new(), + memory_manifest_ref: presetup.memory_manifest_ref, + disk_manifest_ref: presetup.disk_manifest_ref.ok_or_else(|| { + ApiError::Internal("live export has no disk manifest".into()) + })?, + new_memory_chunk_hashes: Vec::new(), + new_disk_chunk_hashes: Vec::new(), + hot_chunks: presetup.hot_chunks, + post_copy: true, + peer_addr: Some(peer_addr), + peer_token: Some(presetup.peer_token), + sidecar_json: presetup.sidecar_json, + }), + source_sandbox_id: None, + state_blob_key: None, + sidecar_blob_key: None, + rootfs_blob_key: None, + working_set_blob_key: None, + aux_bundles: durable.map(|s| s.aux_bundles).unwrap_or_default(), + paused_at: None, + peer_hints: Vec::new(), + }; + return advance( + ctx, + row, + TeleportPhase::Admitted, + TeleportPatch { + export_id: Some(presetup.export_id), + live_payload: Some( + serde_json::to_value(metadata) + .map_err(|e| ApiError::Internal(e.to_string()))?, + ), + ..Default::default() + }, + ) + .await; + } + let metadata = live_metadata(ctx, row).await?; + let dest = state.host_registry.backend_for(row.dest_host_id).await?; + let fence = ctx.fence(); + let mut task = tokio::task::JoinSet::new(); + task.spawn(async move { dest.restore(metadata, fence).await }); + let captured = host + .migration_capture_postcopy( + row.source_sandbox_id, + row.export_id.as_deref().expect("export set"), + ctx.fence(), + ) + .await; + if let Err(e) = captured { + task.abort_all(); + let error = if matches!(e, SandboxError::NotFound | SandboxError::AlreadyExists) + || e.to_string().contains("presetup") + { + "live_export_lost".into() + } else { + e.to_string() + }; + return rollback_begin(ctx, row, error).await; + } + if let Some(out) = + advance(ctx, row, TeleportPhase::Captured, TeleportPatch::default()).await? + { + task.abort_all(); + return Ok(Some(out)); + } + let mut captured_row = row.clone(); + captured_row.phase = TeleportPhase::Captured; + let restored = task + .join_next() + .await + .expect("restore task") + .map_err(|e| ApiError::Unavailable(e.to_string()))?; + finish_live_restore(ctx, &captured_row, restored).await + } + async fn finish_live_restore( + ctx: &OpCtx<'_>, + row: &TeleportRow, + result: Result, + ) -> Result, ApiError> { + match result { + Ok(id) => { + ctx.state + .host_registry + .record_sandbox_owner(id, row.dest_host_id); + advance( + ctx, + row, + TeleportPhase::Restored, + TeleportPatch { + dest_sandbox_id: Some(id), + ..Default::default() + }, + ) + .await + } + Err(SandboxError::NotFound | SandboxError::AlreadyExists) => { + rollback_begin(ctx, row, "live_export_lost".into()).await + } + Err(e) if e.to_string().contains("postcopy-never-loaded") => { + ctx.state + .host_registry + .backend_for(row.source_host_id) + .await? + .migration_abort( + row.source_sandbox_id, + row.export_id.as_deref().expect("live export"), + ctx.fence(), + ) + .await?; + rollback_begin(ctx, row, e.to_string()).await + } + Err(e) if ctx.op.attempts <= 3 => Ok(Some(OpOutcome::Retry(e.to_string()))), + Err(e) => fail_move(ctx, row, &format!("dest_lost_after_blackout: {e}"), false).await, + } + } + async fn destination_gone(ctx: &OpCtx<'_>, row: &TeleportRow) -> Result { + Ok(ctx + .state + .services + .meta + .get_host(row.dest_host_id) + .await? + .is_none_or(|h| matches!(h.status, HostStatus::Dead | HostStatus::Retired))) + } + async fn destroy_destination(ctx: &OpCtx<'_>, row: &TeleportRow) -> Result<(), ApiError> { + if let Some(sandbox) = row.dest_sandbox_id { + if !destination_gone(ctx, row).await? { + match ctx + .state + .host_registry + .backend_for(row.dest_host_id) + .await? + .destroy(sandbox, ctx.fence()) + .await + { + Ok(()) | Err(SandboxError::NotFound) => {} + Err(e) => return Err(e.into()), + } + } + ctx.state + .services + .meta + .record_sandbox_tombstone(row.dest_host_id, sandbox, Some(row.session_id)) + .await?; + } + Ok(()) + } + async fn fail_move( + ctx: &OpCtx<'_>, + row: &TeleportRow, + error: &str, + allow_snapshot: bool, + ) -> Result, ApiError> { + let state = ctx.state; + destroy_destination(ctx, row).await?; + let session = state.services.meta.get_session(row.session_id).await?; + // Release runs after attach. Return through Evacuating before detaching. + if session.status == SessionState::Active { + crate::session_ops::transition_with_fence( + state, + row.session_id, + ctx.fence(), + SessionState::Evacuating, + BindingDisposition::Retain, + ) + .await?; + } + let target = settle_target(ctx, row, &session, allow_snapshot).await?; + // The source tombstone, the session's terminal state, and the + // failed row land in ONE fenced transaction: a driver that lost the + // lane writes none of them. + state + .services + .meta + .teleport_settle( + row.id, + ctx.epoch, + TeleportSettle { + error: error.into(), + session: Some(target), + entomb_source: true, + }, + ) + .await?; + Ok(Some(OpOutcome::Done)) + } + /// Where a session rests when its move fails: the dead-host predicate + /// (`recovery_target`: Idle with a recoverable memory snapshot or a live + /// disk manifest, Dead otherwise), except that a session parked at + /// Created by a deterministic spawn failure has no Dead edge and rests + /// at Failed. `allow_snapshot = false` discards the snapshot evidence + /// (the move itself produced it and it is not trusted). + async fn settle_target( + ctx: &OpCtx<'_>, + row: &TeleportRow, + session: &engram_core::types::Session, + allow_snapshot: bool, + ) -> Result { + let has_recoverable_snapshot = allow_snapshot + && ctx + .state + .services + .meta + .latest_snapshot_for_session(row.session_id) + .await? + .is_some_and(|s| s.recoverable); + let target = crate::dead_host::recovery_target( + has_recoverable_snapshot, + session.live_disk_manifest.is_some(), + ); + Ok(match (session.status, target) { + (SessionState::Created, SessionState::Dead) => SessionState::Failed, + (_, t) => t, + }) + } + + pub(super) async fn admit( + ctx: &OpCtx<'_>, + target: Option, + reason: TeleportReason, + ) -> Result { + let state = ctx.state; + let session = state.services.meta.get_session(ctx.op.session_id).await?; + if session.status != SessionState::Active { + return Ok(TeleportAdmitOutcome::SessionNotActive(session.status)); + } + let (mem, cpu) = + crate::boot_materializer::resolve_resume_budget(&state.services.meta, &session) + .await + .ok_or_else(|| ApiError::Unavailable("teleport budget unavailable".into()))?; + let caps = crate::placement::CapabilityRequirements { + needs_uffd_substrate: true, + fc_snapshot_version: match session.host_id { + Some(h) => state.services.meta.fc_snapshot_version_for_host(h).await?, + None => None, + }, + }; + if let Some(target) = target { + let host = state + .services + .meta + .get_host(target) + .await? + .ok_or_else(|| ApiError::Conflict("no_fit".into()))?; + if host.cordoned { + return Err(ApiError::Conflict("target_cordoned".into())); + } + if crate::placement::host_meets_capabilities(&host, &caps).is_err() { + return Err(ApiError::Conflict("target_lacks_capability".into())); + } + } + let context = crate::placement::ScheduleContext { + repo: &session.image, + image_version: "", + snapshot_host: None, + memory_mib: Some(mem), + cpu_budget_vcpus: Some(cpu), + required_image_digest: None, + exclude_host: session.host_id, + prefer_host: None, + caps, + prefer_bundles: &[], + }; + let candidates = crate::placement::candidates_for( + state.services.meta.as_ref(), + &context, + state.services.clock.now_utc(), + ) + .await + .map_err(|e| ApiError::Unavailable(format!("placement: {e:?}")))? + .hosts; + if target.is_some_and(|h| !candidates.contains(&h)) { + return Ok(TeleportAdmitOutcome::NoFit); + } + let live_host = |h: &engram_core::types::host::HostRecord| { + h.capabilities.backend == "firecracker" + && matches!( + h.capabilities.base_shm_tmpfs, + engram_core::types::host::CapStatus::Ok(_) + ) + && matches!( + h.capabilities.uffd_minor_shmem, + engram_core::types::host::CapStatus::Ok(_) + ) + && matches!( + h.capabilities.nbd, + engram_core::types::host::CapStatus::Ok(_) + ) + }; + let hosts = state.services.meta.list_active_hosts().await?; + let live_capable = session + .host_id + .is_some_and(|source| hosts.iter().any(|h| h.id == source && live_host(h))) + && candidates + .iter() + .filter(|id| target.is_none_or(|t| t == **id)) + .all(|id| hosts.iter().any(|h| h.id == *id && live_host(h))); + state + .services + .meta + .teleport_admit(TeleportAdmitRequest { + id: engram_core::TeleportId::from(state.services.entropy.uuid()), + session_id: session.id, + reason, + epoch: ctx.epoch, + candidates, + pinned_dest: target, + mem_budget_mib: i64::from(mem), + cpu_budget_vcpus: i64::from(cpu), + max_open_per_dest: ctx + .op + .payload + .get("max_open_per_dest") + .and_then(|n| n.as_u64()) + .and_then(|n| u32::try_from(n).ok()) + .unwrap_or(TeleportConfig::default().max_open_per_dest), + live_capable, + }) + .await + .map_err(Into::into) + } + pub(super) async fn capture( + ctx: &OpCtx<'_>, + row: &TeleportRow, + ) -> Result, ApiError> { + if row.kind == TeleportKind::Live { + return capture_live(ctx, row).await; + } + let host = match ctx + .state + .host_registry + .backend_for(row.source_host_id) + .await + { + Ok(host) => host, + Err(e) => return rollback_begin(ctx, row, e.to_string()).await, + }; + let captured = host.snapshot_hold(row.source_sandbox_id, ctx.fence()).await; + let metadata = match captured { + Ok(m) => m, + Err(e) => return rollback_begin(ctx, row, e.to_string()).await, + }; + let record = SnapshotRecord { + id: metadata.id, + session_id: Some(row.session_id), + host_id: Some(row.source_host_id), + image_version: metadata.image_version.clone(), + size_bytes: metadata.size_bytes, + created_at: metadata.created_at, + last_accessed_at: ctx.state.services.clock.now_utc(), + disk_manifest: metadata.disk_manifest, + memory_manifest: metadata.memory_manifest, + recoverable: crate::api::snapshot::verify_snapshot_recoverable( + ctx.state.services.blob.as_ref(), + metadata.disk_manifest.as_ref(), + metadata.memory_manifest.as_ref(), + ) + .await, + aux_bundles: metadata.aux_bundles, + events_cursor: None, + fc_snapshot_version: ctx + .state + .services + .meta + .fc_snapshot_version_for_host(row.source_host_id) + .await?, + }; + if !ctx + .state + .services + .meta + .fenced_record_snapshot(record, ctx.epoch) + .await? + { + return Ok(Some(OpOutcome::Done)); + } + advance( + ctx, + row, + TeleportPhase::Captured, + TeleportPatch { + snapshot_id: Some(metadata.id), + ..Default::default() + }, + ) + .await + } + pub(super) async fn restore( + ctx: &OpCtx<'_>, + row: &TeleportRow, + ) -> Result, ApiError> { + if row.kind == TeleportKind::Live { + let metadata = live_metadata(ctx, row).await?; + let result = ctx + .state + .host_registry + .backend_for(row.dest_host_id) + .await? + .restore(metadata, ctx.fence()) + .await; + return finish_live_restore(ctx, row, result).await; + } + let state = ctx.state; + let snapshot = + state + .services + .meta + .get_snapshot(row.snapshot_id.ok_or_else(|| { + ApiError::Internal("captured teleport has no snapshot".into()) + })?) + .await? + .ok_or_else(|| ApiError::Internal("teleport snapshot missing".into()))?; + let session = state.services.meta.get_session(row.session_id).await?; + let peer_hints = state + .services + .meta + .get_host(row.source_host_id) + .await? + .and_then(|h| { + crate::placement::host_can_serve_chunks( + &h, + state.services.clock.now_utc(), + crate::placement::placement_ttl(), + ) + .map(str::to_owned) + }) + .into_iter() + .collect(); + let metadata = SnapshotMetadata { + id: snapshot.id, + size_bytes: snapshot.size_bytes, + created_at: snapshot.created_at, + image_version: snapshot.image_version, + disk_manifest: snapshot.disk_manifest, + memory_manifest: snapshot.memory_manifest, + base_memory_manifest: crate::api::snapshot::base_memory_manifest_for_image( + state, + &session.image, + ) + .await, + migration_source: None, + source_sandbox_id: None, + state_blob_key: Some(engram_chunk_store::snapshot_blob::state_blob_key( + snapshot.id, + )), + sidecar_blob_key: Some(engram_chunk_store::snapshot_blob::sidecar_blob_key( + snapshot.id, + )), + rootfs_blob_key: None, + working_set_blob_key: None, + aux_bundles: snapshot.aux_bundles, + paused_at: None, + peer_hints, + }; + let host = match state.host_registry.backend_for(row.dest_host_id).await { + Ok(host) => host, + Err(e) => return rollback_begin(ctx, row, e.to_string()).await, + }; + match host.restore(metadata, ctx.fence()).await { + Ok(id) => { + state + .host_registry + .record_sandbox_owner(id, row.dest_host_id); + advance( + ctx, + row, + TeleportPhase::Restored, + TeleportPatch { + dest_sandbox_id: Some(id), + ..Default::default() + }, + ) + .await + } + Err(e) => rollback_begin(ctx, row, e.to_string()).await, + } + } + pub(super) async fn commit( + ctx: &OpCtx<'_>, + row: &TeleportRow, + ) -> Result, ApiError> { + match ctx + .state + .services + .meta + .teleport_commit(row.id, ctx.epoch) + .await? + { + Some(_) => { + ctx.state + .host_registry + .invalidate_sandbox(row.source_sandbox_id); + Ok(None) + } + None => rollback_begin(ctx, row, "commit_conflict".into()).await, + } + } + pub(super) async fn attach( + ctx: &OpCtx<'_>, + row: &TeleportRow, + ) -> Result, ApiError> { + if destination_gone(ctx, row).await? { + return fail_move(ctx, row, "dest_lost_after_blackout", true).await; + } + let state = ctx.state; + let session = state.services.meta.get_session(row.session_id).await?; + let sandbox = row.dest_sandbox_id.ok_or_else(|| { + ApiError::Internal("committed teleport has no destination sandbox".into()) + })?; + // A restart after the state transition still has to release the source. + if matches!(session.status, SessionState::Active | SessionState::Created) + && session.sandbox_id == Some(sandbox) + { + return advance(ctx, row, TeleportPhase::Attached, TeleportPatch::default()).await; + } + let (epoch, attached) = state + .services + .meta + .session_binding_generations(row.session_id) + .await?; + crate::api::snapshot::bind_harness_generation(state, row.session_id, sandbox, epoch) + .await?; + let plan = + crate::boot_materializer::materialize_snapshot_resume(state, &session, sandbox, epoch) + .await?; + attach_plan(ctx, row, plan, epoch, attached).await + } + pub(super) async fn attach_plan( + ctx: &OpCtx<'_>, + row: &TeleportRow, + plan: crate::boot_materializer::HarnessPlan, + epoch: u64, + attached: u64, + ) -> Result, ApiError> { + let state = ctx.state; + let sandbox = row.dest_sandbox_id.expect("committed destination"); + if let crate::boot_materializer::HarnessPlan::Spawn { agent, policy } = plan { + let host = state.host_registry.backend_for(row.dest_host_id).await?; + match host.start_agent(sandbox, agent, *policy, ctx.fence()).await { + Ok(()) => {} + // The same classification ordinary resume uses: a guest + // that cannot spawn this harness today will not spawn it + // on retry, so the session rests at Created (/exec → 409). + Err(e) + if matches!(&e, SandboxError::InvalidSpec(_)) + || matches!(&e, SandboxError::HarnessSpawn { kind, .. } + if engram_core::harness_spawn_kind_is_deterministic(kind)) => + { + let error = e.to_string(); + crate::session_ops::transition_with_fence_emitting( + state, + row.session_id, + ctx.fence(), + SessionState::Created, + BindingDisposition::Retain, + vec![SessionEvent::StatusChanged { + from: SessionState::Evacuating, + to: SessionState::Created, + at: state.services.clock.now_utc(), + }], + ) + .await?; + return advance( + ctx, + row, + TeleportPhase::Attached, + TeleportPatch { + error: Some(error), + ..Default::default() + }, + ) + .await; + } + Err(e) => return Err(e.into()), + } + if attached < epoch { + return Ok(Some(OpOutcome::RetryAfter( + Duration::from_secs(2), + "waiting for harness generation".into(), + ))); + } + } + let mut events = vec![SessionEvent::StatusChanged { + from: SessionState::Evacuating, + to: SessionState::Active, + at: state.services.clock.now_utc(), + }]; + if let Some(snapshot_id) = row.snapshot_id { + events.insert( + 0, + SessionEvent::Resumed { + snapshot_id, + at: state.services.clock.now_utc(), + }, + ); + } + crate::session_ops::transition_with_fence_emitting( + state, + row.session_id, + ctx.fence(), + SessionState::Active, + BindingDisposition::Retain, + events, + ) + .await?; + advance(ctx, row, TeleportPhase::Attached, TeleportPatch::default()).await + } + pub(super) async fn release( + ctx: &OpCtx<'_>, + row: &TeleportRow, + ) -> Result, ApiError> { + if destination_gone(ctx, row).await? { + return fail_move(ctx, row, "dest_lost_after_blackout", true).await; + } + if row.kind == TeleportKind::Live { + let dest = ctx + .state + .host_registry + .backend_for(row.dest_host_id) + .await?; + match dest + .migration_drain_wait(row.dest_sandbox_id.expect("attached destination")) + .await? + { + engram_core::types::snapshot::DrainOutcome::Done { .. } => {} + engram_core::types::snapshot::DrainOutcome::PeerLost { .. } => { + return fail_move(ctx, row, "peer_lost", true).await + } + } + if !source_gone(ctx, row).await? { + match ctx + .state + .host_registry + .backend_for(row.source_host_id) + .await? + .migration_commit( + row.source_sandbox_id, + row.export_id.as_deref().expect("live export"), + ctx.fence(), + ) + .await + { + // Already consumed: an earlier commit passed its point + // of no return, or the host's TTL sweep took it. The + // destroy below confirms the source is gone either way. + Ok(()) | Err(SandboxError::NotFound) => {} + Err(e) => return Err(e.into()), + } + } + } + let result = match ctx + .state + .host_registry + .backend_for(row.source_host_id) + .await + { + Ok(host) => host.destroy(row.source_sandbox_id, ctx.fence()).await, + Err(e) => Err(e), + }; + let how = match result { + Ok(()) | Err(SandboxError::NotFound) => SourceRelease::DestroyAcked, + Err(_) if source_gone(ctx, row).await? => SourceRelease::SourceHostGone, + Err(e) => { + return Ok(Some(OpOutcome::RetryAfter( + Duration::from_secs(10), + e.to_string(), + ))) + } + }; + ctx.state + .services + .meta + .teleport_release_source(row.id, ctx.epoch, how) + .await?; + Ok(Some(OpOutcome::Done)) + } + /// The source is gone during a rollback (dead or retired host, or a + /// destroyed sandbox): entomb it and settle the session the way a lost + /// host settles it (`dead_host::recovery_target`): Idle with a + /// recoverable memory snapshot or a live disk manifest (the disk-only + /// cold boot), Dead with nothing recoverable. + async fn settle_lost_source( + ctx: &OpCtx<'_>, + row: &TeleportRow, + ) -> Result, ApiError> { + let meta = &ctx.state.services.meta; + let session = meta.get_session(row.session_id).await?; + let target = settle_target(ctx, row, &session, true).await?; + meta.teleport_settle( + row.id, + ctx.epoch, + TeleportSettle { + error: "source_lost_during_rollback".into(), + session: Some(target), + entomb_source: true, + }, + ) + .await?; + Ok(Some(OpOutcome::Done)) + } + pub(super) async fn rollback( + ctx: &OpCtx<'_>, + row: &TeleportRow, + ) -> Result, ApiError> { + destroy_destination(ctx, row).await?; + if source_gone(ctx, row).await? { + return settle_lost_source(ctx, row).await; + } + let host = ctx + .state + .host_registry + .backend_for(row.source_host_id) + .await?; + // A live move that reached presetup fenced its export on the source; + // migration_abort unfences it and restores the presetup state. A + // snapshot move (or a live move downgraded before presetup) was held + // paused by snapshot_hold; resume wakes it and re-arms swap. + let resumed = match row.export_id.as_deref() { + Some(export_id) if row.kind == TeleportKind::Live => { + match host + .migration_abort(row.source_sandbox_id, export_id, ctx.fence()) + .await + { + Ok(()) => Ok(()), + // The export is already consumed: an earlier abort passed + // its point of no return, or the host's TTL sweep took it. + // Only the un-pause can still be missing, and resume is + // idempotent on a running guest. Active is declared only + // on a resume ack, never on the consumed export alone. + Err(SandboxError::NotFound) => { + host.resume(row.source_sandbox_id, ctx.fence()).await + } + Err(e) => Err(e), + } + } + _ => host.resume(row.source_sandbox_id, ctx.fence()).await, + }; + match resumed { + Ok(()) => {} + // The source sandbox itself is gone: there is nothing to go back + // to, so the move fails honestly instead of retrying forever. + Err(SandboxError::NotFound) => return settle_lost_source(ctx, row).await, + Err(e) => { + return Ok(Some(OpOutcome::RetryAfter( + Duration::from_secs(5), + e.to_string(), + ))) + } + } + let session = ctx.state.services.meta.get_session(row.session_id).await?; + if session.status == SessionState::Evacuating { + crate::session_ops::transition_with_fence_emitting( + ctx.state, + row.session_id, + ctx.fence(), + SessionState::Active, + BindingDisposition::Retain, + vec![SessionEvent::StatusChanged { + from: SessionState::Evacuating, + to: SessionState::Active, + at: ctx.state.services.clock.now_utc(), + }], + ) + .await?; + } + ctx.state + .services + .meta + .teleport_abort(row.id, ctx.epoch) + .await?; + Ok(Some(OpOutcome::Done)) + } +} + +#[cfg(test)] +use crate as coordinator; +#[cfg(test)] +#[path = "../tests/support/teleport_scenarios.rs"] +mod tests; + +#[cfg(test)] +mod attach_tests { + use super::*; + use std::sync::atomic::Ordering; + use tests::support::Rig; + async fn committed() -> Rig { + let rig = Rig::sim().await; + let meta = &rig.state.services.meta; + let epoch = rig.op.epoch.unwrap(); + meta.teleport_advance( + rig.row.id, + TeleportPhase::Admitted, + TeleportPhase::Captured, + TeleportPatch::default(), + epoch, + ) + .await + .unwrap(); + meta.teleport_advance( + rig.row.id, + TeleportPhase::Captured, + TeleportPhase::Restored, + TeleportPatch { + dest_sandbox_id: Some(rig.dest.sandbox), + ..Default::default() + }, + epoch, + ) + .await + .unwrap(); + meta.teleport_commit(rig.row.id, epoch) + .await + .unwrap() + .unwrap(); + rig.state + .host_registry + .record_sandbox_owner(rig.dest.sandbox, rig.row.dest_host_id); + rig + } + fn plan(rig: &Rig, epoch: u64) -> crate::boot_materializer::HarnessPlan { + crate::boot_materializer::HarnessPlan::Spawn { + agent: engram_core::types::sandbox::AgentSpec { + argv: vec!["harness".into()], + env: Default::default(), + session_env: Default::default(), + binding_epoch: epoch, + host_ca_pem: None, + }, + policy: Box::new(crate::api::snapshot::placeholder_egress_policy( + rig.row.session_id, + rig.dest.sandbox, + )), + } + } + #[tokio::test] + async fn attach_waits_for_the_new_generation_before_active() { + let rig = committed().await; + let meta = &rig.state.services.meta; + let row = meta + .open_teleport_for_session(rig.row.session_id) + .await + .unwrap() + .unwrap(); + let (generation, attached) = meta + .session_binding_generations(row.session_id) + .await + .unwrap(); + let ctx = OpCtx { + state: &rig.state, + op: &rig.op, + epoch: rig.op.epoch.unwrap(), + }; + assert!(matches!( + steps::attach_plan(&ctx, &row, plan(&rig, generation), generation, attached) + .await + .unwrap(), + Some(OpOutcome::RetryAfter(_, _)) + )); + assert_eq!( + meta.get_session(row.session_id).await.unwrap().status, + SessionState::Evacuating + ); + meta.settle_harness_generation( + row.session_id, + generation, + &[], + rig.state.services.clock.now_utc(), + ) + .await + .unwrap(); + let (_, attached) = meta + .session_binding_generations(row.session_id) + .await + .unwrap(); + assert!( + steps::attach_plan(&ctx, &row, plan(&rig, generation), generation, attached) + .await + .unwrap() + .is_none() + ); + assert_eq!( + meta.get_session(row.session_id).await.unwrap().status, + SessionState::Active + ); + } + #[tokio::test] + async fn attach_deterministic_failure_rests_at_created_and_still_releases_source() { + let rig = committed().await; + rig.dest.spawn_fails.store(true, Ordering::SeqCst); + let meta = &rig.state.services.meta; + let row = meta + .open_teleport_for_session(rig.row.session_id) + .await + .unwrap() + .unwrap(); + let (generation, attached) = meta + .session_binding_generations(row.session_id) + .await + .unwrap(); + let ctx = OpCtx { + state: &rig.state, + op: &rig.op, + epoch: rig.op.epoch.unwrap(), + }; + assert!( + steps::attach_plan(&ctx, &row, plan(&rig, generation), generation, attached) + .await + .unwrap() + .is_none() + ); + assert_eq!( + meta.get_session(row.session_id).await.unwrap().status, + SessionState::Created + ); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.source.destroys.load(Ordering::SeqCst), 1); + } +} diff --git a/crates/engram-coordinator/tests/admin_evac_live_pg.rs b/crates/engram-coordinator/tests/admin_evac_live_pg.rs index 4df6b732c..f15056538 100644 --- a/crates/engram-coordinator/tests/admin_evac_live_pg.rs +++ b/crates/engram-coordinator/tests/admin_evac_live_pg.rs @@ -1,30 +1,4 @@ -//! Live-Postgres integration tests for ADR 0018 session evacuation. -//! -//! Exercises `evacuate_dead_source` (dead/degraded source) against a -//! real PG + the in-memory HostRegistry + mock HostClient backends. -//! The PG side is what these tests really exercise — the rebind of -//! `sessions.host_id` / `sessions.sandbox_id` via the -//! `assign_session_*` and `transition_session` calls must work -//! against the real legality-table-enforcing implementation, not a -//! mock. -//! -//! `#[ignore]`'d by default; requires Postgres at -//! `ENGRAM_TEST_DATABASE_URL`. CI wires this into the -//! Postgres-gated-ignored lane alongside `admin_chunk_gc_live_pg`. -//! -//! Coverage: -//! - `evacuate_dead_source_with_snapshot_uses_recorded_manifests` -//! — restores from a recorded snapshot's disk+memory manifests -//! when only the snapshot is available. Loss=None. -//! - `evacuate_dead_source_disk_only_records_memory_loss` — when -//! only `sessions.live_disk_manifest_*` is set (no snapshot row), -//! the receipt carries `EvacLoss::Memory{reason: "source-dead-..."}`. -//! - `evacuate_dead_source_no_state_returns_no_recoverable` — both -//! manifests absent → typed error; PG row stays at HostLost so the -//! caller (the `evac_resumer` scanner, fed by operator drain) can -//! surface it. (As of ADR 0045 Phase A the dead-host detector no -//! longer calls this — it routes recoverable sessions to Idle and -//! the rest to Dead directly; this primitive is drain-only now.) +//! Host cordon, lease, and binding-strike tests against Postgres. // tests drive a live system; wall clock/OS entropy here is input, not a decision source (ADR 0098 D1) #![allow(clippy::disallowed_methods)] @@ -34,33 +8,21 @@ use std::sync::Arc; use async_trait::async_trait; use chrono::Utc; -use engram_chunk_store::{ChunkStore, ManifestKind, ManifestRef as ChunkManifestRef}; -use engram_coordinator::evacuation::{evacuate_dead_source, EvacError}; /// ADR 0116: the lazily-passed cold-boot materialization, pre-resolved /// for tests (production passes `materialize_cold_boot` un-awaited). -fn ready_spec( - spec: Option, -) -> impl std::future::Future, engram_coordinator::error::ApiError>> -{ - std::future::ready(Ok(spec)) -} -use engram_coordinator::host_registry::HostRegistry; use engram_core::traits::{HarnessDial, HostClient, MetadataStore}; -use engram_core::types::evacuation::EvacLoss; -use engram_core::types::manifest::ManifestRef; + use engram_core::types::sandbox::{ExecRequest, ExecStream, SandboxSpec}; use engram_core::types::session::{SessionMode, SessionSpec, SessionState}; -use engram_core::types::snapshot::{SnapshotMetadata, SnapshotRecord}; +use engram_core::types::snapshot::SnapshotMetadata; use engram_core::{HostId, SandboxError, SandboxId, SessionId, SnapshotId}; use engram_storage_local::LocalBlobStorage; use parking_lot::Mutex; struct TestRig { meta: Arc, - chunk_store: ChunkStore, /// URL of this rig's private database, for tests that need a raw pool. - db_url: String, _blob_dir: tempfile::TempDir, } @@ -70,19 +32,16 @@ async fn rig() -> Option { // would be a legal pick. ADR 0099 H1: each rig clones its own // database from the migrated template. let db = engram_testkit::pg::fresh_db().await?; - let db_url = db.url; let store = db.store; let blob_dir = tempfile::tempdir().expect("tempdir"); let blob: Arc = Arc::new(LocalBlobStorage::new(blob_dir.path().to_path_buf())); - let chunk_store = ChunkStore::new(blob); + let _ = blob; let pg = Arc::new(store); Some(TestRig { meta: pg as Arc, - chunk_store, - db_url, _blob_dir: blob_dir, }) } @@ -101,9 +60,6 @@ impl FakeBackend { fn new() -> Arc { Arc::new(Self::default()) } - fn set_restore_id(&self, id: SandboxId) { - *self.next_restore_id.lock() = Some(id); - } } #[async_trait] @@ -346,366 +302,6 @@ async fn seed_active_session( session_id } -async fn seed_manifest(store: &ChunkStore, kind: ManifestKind, label: &str) -> ChunkManifestRef { - use engram_chunk_store::{ChunkRef, Manifest}; - let bytes = format!("{label}-{}", uuid::Uuid::new_v4()).into_bytes(); - let hash = store.put_chunk(&bytes).await.expect("put chunk"); - let mut manifest = Manifest::empty(kind, bytes.len() as u64); - manifest.chunks.push(ChunkRef { offset: 0, hash }); - let r = ChunkManifestRef::new(); - store - .put_manifest(r, &manifest) - .await - .expect("put manifest"); - r -} - -fn proto_to_core_manifest(r: ChunkManifestRef) -> ManifestRef { - ManifestRef { - manifest_id: r.manifest_id, - version: r.version, - } -} - -/// ADR 0028 Fix B: the cold-boot spec a disk-only recovery rides (in -/// prod, derived from the enabled image via `materialize_cold_boot`). -fn test_cold_boot_spec() -> SandboxSpec { - SandboxSpec { - image: "ghcr.io/test/img:t".into(), - rootfs_source: None, - image_uri: Some("ghcr.io/test/img:t".into()), - rootfs_manifest: None, - cpu: engram_core::types::sandbox::CpuLimit { vcpus: 2 }, - memory: engram_core::types::sandbox::MemoryLimit { max_mib: 4096 }, - disk: engram_core::types::sandbox::DiskLimit { max_gib: 20 }, - ttl: None, - env: Default::default(), - workdir: None, - network: Default::default(), - aux_ro_drives: Vec::new(), - swap_mib: None, - } -} - -#[tokio::test] -#[ignore = "requires live Postgres at ENGRAM_TEST_DATABASE_URL"] -async fn evacuate_dead_source_with_snapshot_uses_recorded_manifests() { - let Some(rig) = rig().await else { return }; - let meta = rig.meta.clone(); - let registry = Arc::new(HostRegistry::new(meta.clone())); - - let dead_source = HostId::new(); - let target_host = HostId::new(); - let target_be = FakeBackend::new(); - let new_sandbox = SandboxId::new(); - target_be.set_restore_id(new_sandbox); - registry.register(target_host, target_be.clone()); - - let session_id = seed_active_session(&meta, dead_source, SandboxId::new()).await; - ensure_host_row(&meta, target_host, "target").await; - // Simulate dead_host.rs's first-stage flip: Active → HostLost. - meta.transition_session( - session_id, - SessionState::HostLost, - BindingDisposition::Retain, - ) - .await - .expect("Active → HostLost"); - - // Record a recoverable snapshot for this session. - let disk = seed_manifest(&rig.chunk_store, ManifestKind::Disk, "snap-disk").await; - let memory = seed_manifest(&rig.chunk_store, ManifestKind::Memory, "snap-mem").await; - meta.record_snapshot(SnapshotRecord { - id: SnapshotId::new(), - session_id: Some(session_id), - host_id: None, - image_version: "test".into(), - size_bytes: 1024, - created_at: Utc::now(), - last_accessed_at: Utc::now(), - disk_manifest: Some(proto_to_core_manifest(disk)), - memory_manifest: Some(proto_to_core_manifest(memory)), - recoverable: true, - aux_bundles: vec![], - events_cursor: None, - fc_snapshot_version: None, - }) - .await - .expect("record snapshot"); - - let session = meta.get_session(session_id).await.expect("get session"); - let snapshot = meta - .latest_snapshot_for_session(session_id) - .await - .expect("latest_snapshot lookup") - .expect("snapshot present"); - - let receipt = evacuate_dead_source( - ®istry, - &meta, - session, - Some(snapshot), - ready_spec(None), - None, - None, - engram_core::traits::SessionFence::unfenced(), - None, // #800: budget — these tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await - .expect("dead-source evac succeeds"); - assert_eq!(receipt.new_host_id, target_host); - assert_eq!(receipt.new_sandbox_id, new_sandbox); - assert_eq!(receipt.loss, EvacLoss::None); - - let after = meta.get_session(session_id).await.expect("get session"); - assert_eq!(after.status, SessionState::Created); - assert_eq!(after.host_id, Some(target_host)); - assert_eq!(after.sandbox_id, Some(new_sandbox)); -} - -#[tokio::test] -#[ignore = "requires live Postgres at ENGRAM_TEST_DATABASE_URL"] -async fn evacuate_dead_source_disk_only_records_memory_loss() { - let Some(rig) = rig().await else { return }; - let meta = rig.meta.clone(); - let registry = Arc::new(HostRegistry::new(meta.clone())); - - let dead_source = HostId::new(); - let target_host = HostId::new(); - let target_be = FakeBackend::new(); - target_be.set_restore_id(SandboxId::new()); - registry.register(target_host, target_be); - - let old_sandbox = SandboxId::new(); - let session_id = seed_active_session(&meta, dead_source, old_sandbox).await; - ensure_host_row(&meta, target_host, "target").await; - - // Set live_disk_manifest_*. The update is sandbox-id-gated so we - // pass the current binding. - let live_disk = seed_manifest(&rig.chunk_store, ManifestKind::Disk, "live-disk").await; - meta.update_live_disk_manifest(session_id, old_sandbox, proto_to_core_manifest(live_disk)) - .await - .expect("update_live_disk_manifest"); - - meta.transition_session( - session_id, - SessionState::HostLost, - BindingDisposition::Retain, - ) - .await - .expect("Active → HostLost"); - - let session = meta.get_session(session_id).await.expect("get session"); - // ADR 0028 Fix B: disk-only recovery is a cold boot — the caller - // supplies the boot spec (in prod, derived from the enabled image - // via `materialize_cold_boot`). - let receipt = evacuate_dead_source( - ®istry, - &meta, - session, - None, - ready_spec(Some(test_cold_boot_spec())), - None, - None, - engram_core::traits::SessionFence::unfenced(), - None, // #800: budget — these tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await - .expect("disk-only evac succeeds"); - match &receipt.loss { - EvacLoss::Memory { reason } => { - assert_eq!(reason, "source-dead-no-snapshot"); - } - other => panic!("expected Memory loss, got {other:?}"), - } - - let after = meta.get_session(session_id).await.expect("get session"); - assert_eq!(after.status, SessionState::Created); -} - -#[tokio::test] -#[ignore = "requires live Postgres at ENGRAM_TEST_DATABASE_URL"] -async fn evacuate_dead_source_no_state_returns_no_recoverable() { - let Some(rig) = rig().await else { return }; - let meta = rig.meta.clone(); - let registry = Arc::new(HostRegistry::new(meta.clone())); - - let dead_source = HostId::new(); - let target_host = HostId::new(); - registry.register(target_host, FakeBackend::new()); - - let session_id = seed_active_session(&meta, dead_source, SandboxId::new()).await; - meta.transition_session( - session_id, - SessionState::HostLost, - BindingDisposition::Retain, - ) - .await - .expect("Active → HostLost"); - - let session = meta.get_session(session_id).await.expect("get session"); - let result = evacuate_dead_source( - ®istry, - &meta, - session, - None, - ready_spec(None), - None, - None, - engram_core::traits::SessionFence::unfenced(), - None, // #800: budget — these tests keep the capacity-soft pick - chrono::Utc::now(), - ) - .await; - assert!(matches!(result, Err(EvacError::NoRecoverableState))); - - // PG row sits at HostLost — the caller routes it to Dead next. - let after = meta.get_session(session_id).await.expect("get session"); - assert_eq!(after.status, SessionState::HostLost); -} - -// --------------------------------------------------------------------- -// ADR 0018 commit 12j — live-PG tests for the new evac_resumer -// scanner primitives. These run in CI's Postgres-gated lane (same -// `ENGRAM_TEST_DATABASE_URL` requirement) and pin the load-bearing -// behaviour of the async-evac state machine. -// --------------------------------------------------------------------- - -/// Migration 0037: `evac_attempts` column exists, defaults to 0, and -/// the partial index `idx_sessions_evacuating` is created. Pins the -/// schema so a future migration that drops/renames either surfaces -/// here, not at runtime when the scanner's sweep query 500s. -#[tokio::test] -#[ignore = "requires live Postgres at ENGRAM_TEST_DATABASE_URL"] -async fn migration_0037_landed_evac_attempts_column_and_index() { - let Some(rig) = rig().await else { return }; - // Inspect the rig's OWN database — the schema assertions must run - // against what the template migration chain produced, not whatever - // state the shared admin database happens to be in. - let pool = sqlx::PgPool::connect(&rig.db_url).await.unwrap(); - - let row: Option<(String, String, Option)> = sqlx::query_as( - "SELECT column_name, data_type, column_default \ - FROM information_schema.columns \ - WHERE table_name = 'sessions' AND column_name = 'evac_attempts'", - ) - .fetch_optional(&pool) - .await - .unwrap(); - let (col, ty, default) = row.expect("evac_attempts column must exist after migration 0037"); - assert_eq!(col, "evac_attempts"); - assert_eq!(ty, "integer"); - assert!( - default.as_deref().unwrap_or("").starts_with('0'), - "evac_attempts default must be 0, got {default:?}" - ); - - // Verify the CHECK constraint accepts 'evacuating' — without - // this an attempt to UPDATE sessions SET status='evacuating' - // 500s with a constraint violation (caught on dev-vm; that - // failure mode is exactly what the migration's first ALTER - // block guards against). - let idx: Option<(String,)> = sqlx::query_as( - "SELECT indexname FROM pg_indexes WHERE indexname = 'idx_sessions_evacuating'", - ) - .fetch_optional(&pool) - .await - .unwrap(); - assert!(idx.is_some(), "idx_sessions_evacuating must exist"); - - drop(rig); -} - -/// `list_evacuating_sessions` returns the right set + their -/// `evac_attempts` values; `bump_evac_attempts` is atomic +1 -/// RETURNING; `transition_session(Evacuating)` resets the counter. -/// All three are load-bearing for the scanner's sweep loop. -#[tokio::test] -#[ignore = "requires live Postgres at ENGRAM_TEST_DATABASE_URL"] -async fn evac_attempts_primitives_round_trip() { - let Some(rig) = rig().await else { return }; - let meta = rig.meta.clone(); - - let host_id = HostId::new(); - let sandbox_id = SandboxId::new(); - let session_id = seed_active_session(&meta, host_id, sandbox_id).await; - - // Active → Evacuating (legal, sets counter to 0) - meta.transition_session( - session_id, - SessionState::Evacuating, - BindingDisposition::Retain, - ) - .await - .expect("Active → Evacuating"); - - let candidates = meta - .list_evacuating_sessions() - .await - .expect("list_evacuating_sessions"); - let entry = candidates - .iter() - .find(|(s, _)| s.id == session_id) - .expect("our session in the candidate list"); - assert_eq!(entry.1, 0, "counter starts at 0 after Evacuating entry"); - - // bump returns new value; second bump returns 2. - let v1 = meta.bump_evac_attempts(session_id).await.unwrap(); - let v2 = meta.bump_evac_attempts(session_id).await.unwrap(); - assert_eq!(v1, 1); - assert_eq!(v2, 2); - - // list reflects the latest bump. - let candidates = meta.list_evacuating_sessions().await.unwrap(); - let after_bump = candidates.iter().find(|(s, _)| s.id == session_id).unwrap(); - assert_eq!(after_bump.1, 2, "list reads back the bumped count"); - - // Transition out (Evacuating → Idle) — counter NOT reset (only - // re-entry into Evacuating resets, per migration 0037's CASE). - meta.transition_session(session_id, SessionState::Idle, BindingDisposition::Detach) - .await - .expect("Evacuating → Idle (budget-exhaustion fallback shape)"); - let candidates = meta.list_evacuating_sessions().await.unwrap(); - assert!( - candidates.iter().all(|(s, _)| s.id != session_id), - "session no longer Evacuating should drop out of the sweep" - ); - - // Re-enter Evacuating from Idle. Wait — Idle → Evacuating isn't - // legal directly. The supported re-entry path goes through - // Active. Drive Idle → Created → Active → Evacuating to exercise - // the legality + the counter-reset on entry. - meta.assign_session_sandbox(session_id, Some(SandboxId::new())) - .await - .unwrap(); - meta.transition_session( - session_id, - SessionState::Created, - BindingDisposition::Retain, - ) - .await - .expect("Idle → Created"); - meta.transition_session(session_id, SessionState::Active, BindingDisposition::Retain) - .await - .expect("Created → Active"); - meta.transition_session( - session_id, - SessionState::Evacuating, - BindingDisposition::Retain, - ) - .await - .expect("Active → Evacuating (second drain)"); - - let candidates = meta.list_evacuating_sessions().await.unwrap(); - let on_reentry = candidates.iter().find(|(s, _)| s.id == session_id).unwrap(); - assert_eq!( - on_reentry.1, 0, - "re-entry into Evacuating must reset evac_attempts to 0" - ); -} - /// Issue #215 (live-PG): the actual `sessions.missing_strikes` SQL must /// reset on a sandbox re-key. Accrue `grace_ticks - 1` strikes against /// SB1 via `apply_missing_sandbox_strikes`, then rebind to SB2 via the @@ -807,8 +403,8 @@ async fn missing_strikes_reset_on_sandbox_rekey() { #[tokio::test] #[ignore = "requires live Postgres at ENGRAM_TEST_DATABASE_URL"] async fn durable_cordon_excludes_host_from_placement_on_every_replica() { - use engram_coordinator::host_registry::HostRegistry; use engram_coordinator::placement::{self, ScheduleContext}; + use engram_coordinator::HostRegistry; let Some(rig) = rig().await else { return }; let meta = rig.meta.clone(); let registry = Arc::new(HostRegistry::new(meta.clone())); @@ -1159,109 +755,3 @@ async fn delete_host_refuses_unretired_then_idempotent() { other => panic!("a second delete of a gone row must be Deleted, got {other:?}"), } } - -/// ADR 0048 C8 (drain don't-strand guard): `placement_preview` is the -/// HARD 2D fit check `drain_host` runs before starting ANY move. If the -/// only survivor (the victim excluded) can't hold the session's budgets, -/// it returns `false` so the drain surfaces a failure instead of parking -/// an Active session Idle on a full fleet. A measured-but-too-small -/// survivor → false; growing it (or its CPU budget) → true. An UNMEASURED -/// survivor (allocatable 0) keeps the soft-fits posture → true. -#[tokio::test] -#[ignore = "requires live Postgres at ENGRAM_TEST_DATABASE_URL"] -async fn drain_dont_strand_guard_blocks_when_no_survivor_fits() { - use engram_coordinator::placement::{self, ScheduleContext}; - let Some(rig) = rig().await else { return }; - let meta = rig.meta.clone(); - - let victim = HostId::new(); - let survivor = HostId::new(); - seed_ready_host(&meta, victim, "drain-victim").await; - seed_ready_host(&meta, survivor, "drain-survivor").await; - - // Heartbeat the survivor as MEASURED with a small allocatable + a - // CPU budget. allocatable 4096 MiB; total_vcpus 4 ⇒ CPU budget - // 4 × overcommit (default 4.0) = 16 vCPU. - let heartbeat = |alloc_mib: u64, vcpus: u32| { - let meta = meta.clone(); - async move { - meta.touch_host_heartbeat( - survivor, - engram_core::types::host::HostHeartbeat { - status: engram_core::types::HostStatus::Ready, - capacity: engram_core::types::HostCapacity { - total_gb: 0, - used_gb: 0, - total_mib: 65_536, - used_mib: 0, - running_sandboxes: 0, - }, - utilization: engram_core::types::host::HostUtilization { - allocatable_mib: alloc_mib, - ..Default::default() - }, - ready_images: Vec::new(), - current_bundles: Vec::new(), - sandbox_bundles: Vec::new(), - total_vcpus: vcpus, - wire_version: engram_protocol::WIRE_VERSION, - stages_images: false, - capabilities: engram_core::types::host::HostCapabilities::default(), - lease_renew_until: None, - }, - ) - .await - .expect("heartbeat survivor"); - } - }; - heartbeat(4_096, 4).await; - - // The victim is excluded (it's draining); the survivor is the only - // candidate left. - let ctx = ScheduleContext { - repo: "test/img", - image_version: "v1", - snapshot_host: None, - memory_mib: Some(8_192), - cpu_budget_vcpus: Some(2), - required_image_digest: None, - exclude_host: Some(victim), - prefer_host: None, - caps: Default::default(), - prefer_bundles: &[], - }; - - // 8 GiB session, survivor has 4 GiB free → no fit → would strand. - let fits = placement::placement_preview(meta.as_ref(), &ctx, 8_192, 2, chrono::Utc::now()) - .await - .expect("placement_preview"); - assert!( - !fits, - "a 8 GiB session must NOT fit a 4 GiB survivor — the guard blocks the drain" - ); - - // Grow the survivor's RAM → now it fits both dims. - heartbeat(16_384, 4).await; - let fits = placement::placement_preview(meta.as_ref(), &ctx, 8_192, 2, chrono::Utc::now()) - .await - .expect("placement_preview"); - assert!(fits, "a 8 GiB session fits a 16 GiB survivor"); - - // CPU dimension binds independently: plenty of RAM, but a 32-vCPU - // ask against a 4-core × 4.0 = 16-vCPU budget → no fit. - let fits = placement::placement_preview(meta.as_ref(), &ctx, 8_192, 32, chrono::Utc::now()) - .await - .expect("placement_preview"); - assert!( - !fits, - "CPU budget binds before RAM — a 32-vCPU ask exceeds the 16-vCPU host budget" - ); - - // An UNMEASURED survivor (allocatable 0, no reported cores) keeps the - // soft-fits posture reserve_placement takes for brand-new / dev hosts. - heartbeat(0, 0).await; - let fits = placement::placement_preview(meta.as_ref(), &ctx, 8_192, 32, chrono::Utc::now()) - .await - .expect("placement_preview"); - assert!(fits, "an unmeasured survivor soft-fits any budget"); -} diff --git a/crates/engram-coordinator/tests/binding_writer_inventory.rs b/crates/engram-coordinator/tests/binding_writer_inventory.rs index ba637e855..904ee2fd5 100644 --- a/crates/engram-coordinator/tests/binding_writer_inventory.rs +++ b/crates/engram-coordinator/tests/binding_writer_inventory.rs @@ -27,6 +27,8 @@ const METHODS: &[&str] = &[ "assign_session_sandbox", "assign_session_sandbox_guarded", "rebind_session_guarded", + "teleport_commit", + "teleport_settle", "fenced_assign_sandbox", "mark_host_dead_if_lease_expired", "transition_session", @@ -42,24 +44,23 @@ const METHODS: &[&str] = &[ const DECLARED: &[(&str, &str, usize, &str)] = &[ ("api/host_http", "transition_session", 1, "Active→Unreachable (Retain: dead-guest VM stays owned)"), ("api/sessions", "transition_session", 1, "BootError::Started→Failed (Detach: boot pipeline unbound; idempotent)"), - ("api/snapshot", "fenced_assign_sandbox", 1, "#896 resume gate: fenced clear after confirmed source teardown"), + ("api/snapshot", "fenced_assign_sandbox", 2, "resume gate clears retained source; disk-only recovery binds the fresh sandbox"), ("api/snapshot", "rebind_session_guarded", 3, "resume bind plus Created recovery and in-place reattach mint under the expected binding"), ("api/snapshot", "transition_session", 2, "no-recoverable-state / restore-failed →Dead (Detach authorizes orphan reap)"), ("api/snapshot", "transition_with_fence", 3, "ascent→Active + park→Created + resume-finish→Active (all Retain)"), ("dead_host", "assign_session_sandbox_guarded", 1, "straggler destroy's guarded clear before →Idle/Dead"), ("dead_host", "mark_host_dead_if_lease_expired", 1, "the bulk orphan: every non-terminal on a dead host →HostLost unbound"), ("dead_host", "transition_session", 1, "THE consolidated settle_host_lost: HostLost→Idle|Dead (RequireUnbound; ADR 0116 A4 — one settle, three callers)"), - ("evac_resumer", "fenced_assign_sandbox", 1, "confirmed-teardown clear before the peer restore"), - ("evac_resumer", "transition_session", 2, "exhaustion→Idle (THE authorized Retain residue, #896) + structural →Idle/Dead (RequireUnbound)"), - ("evacuation", "assign_session_sandbox", 1, "evac rebind: the new sandbox before →Created"), - ("evacuation", "transition_session", 1, "→Created post-rebind (Retain)"), ("idle_detector", "transition_session", 1, "Active→Evicting nomination (Retain: VM untouched)"), ("idle_evictor", "fenced_assign_sandbox", 1, "quarantine reap unbind"), ("idle_evictor", "transition_session", 3, "evict/park-reaper exhaustion →HostLost (Retain/RequireUnbound) + Parked→Evicting descent (Retain)"), ("idle_evictor", "transition_with_fence", 5, "park ladder + quarantine/unbound →HostLost + admin evict entry"), - ("idle_evictor", "transition_with_fence_emitting", 1, "THE fused evict flip: Detach into Idle, Retain into Evacuating (#896)"), - ("live_migration", "rebind_session_guarded", 1, "teleport commit: new sandbox CAS'd over the expected old one"), - ("live_migration", "transition_session", 5, "parachute arms (Retain) + walk-back (Retain) + kill →Failed (Detach)"), + ("idle_evictor", "transition_with_fence_emitting", 1, "fused eviction completion detaches into Idle"), + ("teleport", "teleport_commit", 1, "atomic teleport destination binding"), + ("teleport", "transition_with_fence_emitting", 3, "attach->Active, rollback->Active, attach-failed->Created (Retain)"), + ("teleport", "teleport_settle", 3, "failed settle: session Detach + source tombstone + failed row in one fenced txn"), + ("teleport", "transition_with_fence", 1, "post-attach peer loss enters failed recovery"), + ("api/snapshot", "transition_with_fence_emitting", 1, "disk-only cold boot retains the fenced binding"), ("queue_scanner", "transition_session", 4, "Queued→Idle/Failed settles (RequireUnbound: queued rows are unbound)"), ("reconcile", "assign_session_sandbox", 1, "strike-out unbind fallback"), ("reconcile", "assign_session_sandbox_guarded", 1, "strike-out guarded unbind"), diff --git a/crates/engram-coordinator/tests/dead_host_mock.rs b/crates/engram-coordinator/tests/dead_host_mock.rs index d105a7dd3..0d051daa1 100644 --- a/crates/engram-coordinator/tests/dead_host_mock.rs +++ b/crates/engram-coordinator/tests/dead_host_mock.rs @@ -447,3 +447,21 @@ async fn arc_dyn_metadata_store_dispatches_correctly() { let result = meta.mark_host_dead_if_lease_expired(HostId::new()).await; assert!(result.is_ok()); } + +use engram_coordinator as coordinator; +#[path = "support/teleport.rs"] +pub mod teleport_support; + +#[tokio::test] +async fn skips_sessions_with_open_teleport_as_source() { + let rig = teleport_support::Rig::sim().await; + let meta = &rig.state.services.meta; + assert!(meta + .mark_host_dead_if_lease_expired(rig.row.source_host_id) + .await + .unwrap() + .is_empty()); + let session = meta.get_session(rig.row.session_id).await.unwrap(); + assert_eq!(session.status, SessionState::Evacuating); + assert_eq!(session.sandbox_id, Some(rig.source.sandbox)); +} diff --git a/crates/engram-coordinator/tests/e2e_stack.rs b/crates/engram-coordinator/tests/e2e_stack.rs index ed9199d2c..f032d60d0 100644 --- a/crates/engram-coordinator/tests/e2e_stack.rs +++ b/crates/engram-coordinator/tests/e2e_stack.rs @@ -533,20 +533,20 @@ impl Driver { self.get_session(sid).await.status } - /// `FleetService.EvacuateSession` — async evacuation. Returns the full + /// `FleetService.TeleportSession` — async evacuation. Returns the full /// gRPC `Status` on error so the shape test can assert on the code (the /// 404/409/202 distinctions the old HTTP handler returned now map to /// gRPC `NotFound` / `FailedPrecondition` / `Ok`). async fn evacuate( &mut self, sid: SessionId, - ) -> Result { - let req = app::EvacuateSessionRequest { + ) -> Result { + let req = app::TeleportSessionRequest { session_id: sid.to_string(), target_host: None, }; self.fleet - .evacuate_session(req) + .teleport_session(req) .await .map(|r| r.into_inner()) } @@ -2058,7 +2058,7 @@ async fn e2e_update_image_cheap_edit_and_recapture_gate() { /// - The RPC is bound under bearer auth. /// - `NotFound` on a non-existent session id (the old HTTP 404). /// - For an Active session: returns Ok with `status="evacuating"` (the old -/// HTTP 202). The `evac_resumer` scanner drives the session to Active on +/// HTTP 202). The `teleport` scanner drives the session to Active on /// a peer in ≤10s. /// /// Pinned regressions: @@ -2076,39 +2076,44 @@ async fn e2e_update_image_cheap_edit_and_recapture_gate() { #[ignore = "requires ENGRAM_E2E_GRPC_ADDR + a baked demo image; runs in ci.yml's test-e2e-stack lane"] async fn e2e_evac_admin_endpoint_shape() { let mut driver = Driver::from_env().await; - let image = Driver::image_uri(); - - // Case 1: NotFound on a session that doesn't exist. - let bogus = SessionId::new(); let err = driver - .evacuate(bogus) + .evacuate(SessionId::new()) .await - .expect_err("evacuate on unknown session must error"); - assert_eq!( - err.code(), - tonic::Code::NotFound, - "evacuate on unknown session should map to NotFound; got {err:?}", - ); - - // Case 2: live session — async shape. Handler must accept immediately - // after marking the session Evacuating. The scanner picks it up from - // there. - let sid = driver.create_session_none_harness(&image).await; - let resp = driver - .evacuate(sid) + .expect_err("unknown session"); + assert_eq!(err.code(), tonic::Code::NotFound); + let hosts = driver + .fleet + .list_hosts(app::ListHostsRequest {}) .await - .expect("evacuate must succeed on an Active session (async shape)"); - assert_eq!( - resp.status, "evacuating", - "evacuate response status must be \"evacuating\"; got {resp:?}", - ); - assert_eq!( - resp.session_id, - sid.to_string(), - "evacuate response must echo session_id; got {resp:?}", - ); - - driver.delete(sid).await; + .unwrap() + .into_inner() + .hosts; + let host = hosts.first().expect("host").id.clone(); + let report = driver + .fleet + .admin_drain_host(app::AdminDrainHostRequest { + host_id: host.clone(), + }) + .await + .unwrap() + .into_inner(); + assert_eq!(report.host_id, host); + assert!(report + .planned + .iter() + .all(|s| s.parse::().is_ok())); + assert!(report + .descended + .iter() + .all(|s| s.parse::().is_ok())); + driver + .fleet + .uncordon_host(app::UncordonHostRequest { + host_id: host, + owner: "admin".into(), + }) + .await + .unwrap(); } /// Whether this environment is REQUIRED to have ≥2 hosts (the teleport/evac @@ -2124,28 +2129,11 @@ fn two_hosts_required() -> bool { ) } -/// Full-stack two-host teleport via `FleetService.EvacuateSession` — the -/// teleport PRIMITIVE on the app-gRPC surface (ADR 0051). The orchestrator's -/// teleport verb is this RPC; `EvacuateSession` pauses + flushes + snapshots -/// the source sandbox, marks the session Evacuating, and the `evac_resumer` -/// scanner resumes it on a PEER host (any non-source host via the standard -/// policy — the proto's `target_host` is reserved/ignored, so unlike the old -/// REST `/teleport` this does not pin the destination). -/// -/// Flow: create on host A → write a UUID sentinel + sync → `EvacuateSession` -/// → wait for Active on a peer → assert the session LEFT host A and the disk -/// sentinel crossed byte-identical. This is the gRPC replacement for the -/// deleted `e2e_two_host_teleport_preserves_sentinel` (which drove the removed -/// REST route): same disk-fidelity-across-relocation guarantee, minus the -/// destination-pinning assertion the evac primitive intentionally doesn't -/// offer. The single-host async-accept shape stays pinned by -/// `e2e_evac_admin_endpoint_shape` above. -/// -/// Requires a two-host stack. On a single-host lane it skips with a warning -/// unless `ENGRAM_EXPECT_TWO_HOSTS=1`, where <2 hosts is a hard failure. +/// Retire a source, preserve disk data on the destination, then delete the empty host. +/// CI runs this after the pinned-target test because it removes a fixture host. #[tokio::test] #[ignore = "requires ENGRAM_E2E_GRPC_ADDR + a two-host stack (ENGRAM_INTEG_TWO_HOSTS=1); runs in ci.yml's test-e2e-stack teleport variant"] -async fn e2e_two_host_evacuate_preserves_sentinel() { +async fn e2e_retire_host_relocates_and_grants() { let mut driver = Driver::from_env().await; let image = Driver::image_uri(); @@ -2187,18 +2175,37 @@ async fn e2e_two_host_evacuate_preserves_sentinel() { write.stderr, ); - // EvacuateSession: async-accept, then the scanner drives Evacuating → - // Created → Active on a peer. - let resp = driver - .evacuate(sid) + driver + .fleet + .retire_host(app::RetireHostRequest { + host_id: src.clone(), + owner: "admin".into(), + reason: "e2e".into(), + }) .await - .expect("EvacuateSession must succeed on an Active session"); - assert_eq!( - resp.status, "evacuating", - "evacuate response status must be \"evacuating\"; got {resp:?}", - ); - - // The evac_resumer drives Evacuating → Created → Active on the peer. + .expect("retirement accepted"); + let deadline = std::time::Instant::now() + Duration::from_secs(180); + loop { + let host = driver + .fleet + .get_host(app::GetHostRequest { + host_id: src.clone(), + }) + .await + .unwrap() + .into_inner() + .host + .unwrap(); + if host.retirement.is_some_and(|r| !r.retired_at.is_empty()) { + break; + } + assert!( + std::time::Instant::now() < deadline, + "retirement did not receive a grant" + ); + tokio::time::sleep(Duration::from_millis(250)).await; + } + // The durable machine attaches the destination before releasing the source. // Generous deadline: first restore on the peer may cold-fetch chunks. // 180s mirrors the cold-create budget. assert!( @@ -2210,7 +2217,7 @@ async fn e2e_two_host_evacuate_preserves_sentinel() { driver.session_status(sid).await, ); - // Left the source host. EvacuateSession picks any non-source peer, so we + // Left the source host. TeleportSession picks any non-source peer, so we // assert the move happened (session left A) — not which peer it landed on. let after = driver .session_host_id(sid) @@ -2236,4 +2243,82 @@ async fn e2e_two_host_evacuate_preserves_sentinel() { ); driver.delete(sid).await; + driver + .fleet + .delete_host(app::DeleteHostRequest { + host_id: src.clone(), + }) + .await + .unwrap(); + driver + .fleet + .delete_host(app::DeleteHostRequest { host_id: src }) + .await + .unwrap(); +} + +#[tokio::test] +#[ignore = "requires the two-host e2e stack; runs before the retirement test in CI"] +async fn e2e_teleport_session_honors_target_host() { + let mut driver = Driver::from_env().await; + let hosts = driver.list_host_ids().await; + assert!(hosts.len() >= 2, "two-host fixture required"); + let sid = driver + .create_session_none_harness(&Driver::image_uri()) + .await; + let source = driver.session_host_id(sid).await.unwrap(); + let target = hosts.into_iter().find(|h| *h != source).unwrap(); + // A source cannot also be its destination. Admission must leave it Active. + let refused = driver + .fleet + .teleport_session(app::TeleportSessionRequest { + session_id: sid.to_string(), + target_host: Some(source.clone()), + }) + .await + .unwrap_err(); + assert_eq!(refused.code(), tonic::Code::FailedPrecondition); + assert_eq!(driver.session_status(sid).await, "active"); + assert_eq!(driver.session_host_id(sid).await, Some(source)); + let move_row = driver + .fleet + .teleport_session(app::TeleportSessionRequest { + session_id: sid.to_string(), + target_host: Some(target.clone()), + }) + .await + .unwrap() + .into_inner(); + assert_eq!(move_row.dest_host_id, target); + assert!(matches!(move_row.kind.as_str(), "snapshot" | "live")); + let mut events = driver + .sess + .stream_events(app::StreamEventsRequest { + session_id: sid.to_string(), + since: None, + durable_only: true, + }) + .await + .unwrap() + .into_inner(); + let done = tokio::time::timeout(Duration::from_secs(180), async { + loop { + let event = events.message().await.unwrap().expect("event stream ended"); + if event.kind == "teleport_finished" + && payload_str(&event.payload_json, "teleport_id").as_deref() + == Some(&move_row.teleport_id) + { + break event; + } + } + }) + .await + .expect("teleport did not finish"); + assert_eq!( + payload_str(&done.payload_json, "outcome").as_deref(), + Some("done") + ); + assert_eq!(driver.session_status(sid).await, "active"); + assert_eq!(driver.session_host_id(sid).await, Some(target)); + driver.delete(sid).await; } diff --git a/crates/engram-coordinator/tests/ha_listener.rs b/crates/engram-coordinator/tests/ha_listener.rs index c14fa02ec..7399cc76a 100644 --- a/crates/engram-coordinator/tests/ha_listener.rs +++ b/crates/engram-coordinator/tests/ha_listener.rs @@ -597,25 +597,48 @@ async fn cross_replica_scheduling_pins_and_tokens() { .await .expect("create session"); meta_a - .set_teleport_target(session_id, Some(h2)) + .assign_session_host(session_id, Some(h1)) .await - .expect("pin via A"); + .unwrap(); + meta_a + .transition_session_created(session_id, engram_core::SandboxId::new()) + .await + .unwrap(); + meta_a + .transition_session( + session_id, + engram_core::types::SessionState::Active, + engram_core::types::BindingDisposition::Retain, + ) + .await + .unwrap(); + let admitted = meta_a + .teleport_admit(engram_core::types::teleport::TeleportAdmitRequest { + id: engram_core::TeleportId::new(), + session_id, + epoch: 0, + reason: engram_core::types::teleport::TeleportReason::Ui, + candidates: vec![h2], + pinned_dest: Some(h2), + mem_budget_mib: 1, + cpu_budget_vcpus: 1, + max_open_per_dest: 1, + live_capable: false, + }) + .await + .unwrap(); + assert!(matches!( + admitted, + engram_core::types::teleport::TeleportAdmitOutcome::Admitted(_) + )); assert_eq!( meta_b - .get_teleport_target(session_id) + .open_teleport_for_session(session_id) .await - .expect("get via B") - .map(|(h, _set_at)| h), - Some(h2), - "B's scanner must honor A's pin" - ); - meta_b - .set_teleport_target(session_id, None) - .await - .expect("clear via B"); - assert_eq!( - meta_a.get_teleport_target(session_id).await.expect("get"), - None + .unwrap() + .unwrap() + .dest_host_id, + h2 ); // --- 4. broker token sealed via A ⇒ unsealed + equal on B -------- diff --git a/crates/engram-coordinator/tests/session_ops_live_pg.rs b/crates/engram-coordinator/tests/session_ops_live_pg.rs index 4f08d1ef9..6960d1eee 100644 --- a/crates/engram-coordinator/tests/session_ops_live_pg.rs +++ b/crates/engram-coordinator/tests/session_ops_live_pg.rs @@ -1008,3 +1008,32 @@ async fn wake_queued_kind_pulls_not_before_to_now() { "no queued deliver left to wake", ); } + +use engram_coordinator as coordinator; +#[path = "support/teleport.rs"] +pub mod teleport_support; + +#[tokio::test] +#[ignore = "requires live Postgres at ENGRAM_TEST_DATABASE_URL"] +async fn teleport_op_resumes_from_row_after_a_long_executor_gap() { + let Some(db) = engram_testkit::pg::fresh_db().await else { + return; + }; + let clock = engram_sim::ManualClock::new(); + let meta = std::sync::Arc::new(db.store.with_clock(clock.clone())); + let mut rig = teleport_support::Rig::new(meta, clock).await; + // The no-deadline contract is pinned by session_ops::deadline_tests. + rig.clock.advance(std::time::Duration::from_secs(3600)); + rig.reclaim().await; + assert!(matches!( + rig.drive().await, + coordinator::session_ops::OpOutcome::Done + )); + assert_eq!(rig.phase().await, None); + assert_eq!( + rig.source + .captures + .load(std::sync::atomic::Ordering::SeqCst), + 1 + ); +} diff --git a/crates/engram-coordinator/tests/support/teleport.rs b/crates/engram-coordinator/tests/support/teleport.rs new file mode 100644 index 000000000..562d83a1f --- /dev/null +++ b/crates/engram-coordinator/tests/support/teleport.rs @@ -0,0 +1,529 @@ +use super::coordinator; +use async_trait::async_trait; +use coordinator::state::SharedState; +use coordinator::{AppState, CoordinatorConfig, HostRegistry, Services}; +use engram_core::traits::{Clock, Entropy, HarnessDial, HostClient, MetadataStore, SessionFence}; +use engram_core::types::host::{HostCapacity, HostRecord, HostStatus}; +use engram_core::types::sandbox::{ExecRequest, ExecStream, SandboxSpec}; +use engram_core::types::session::{SessionMode, SessionSpec, SessionState}; +use engram_core::types::session_op::{EnqueueOutcome, OpKind, SessionOp}; +use engram_core::types::snapshot::SnapshotMetadata; +use engram_core::types::teleport::*; +use engram_core::{HostId, SandboxError, SandboxId, SessionId, SnapshotId}; +use std::sync::{ + atomic::{AtomicBool, AtomicUsize, Ordering}, + Arc, +}; + +pub struct ScriptedHost { + pub sandbox: SandboxId, + pub snapshot: SnapshotId, + pub clock: Arc, + pub capture_fails: AtomicBool, + pub restore_fails: AtomicBool, + pub resume_fails: AtomicBool, + /// `resume` answers NotFound: the source sandbox no longer exists. + pub resume_not_found: AtomicBool, + /// `migration_abort` answers NotFound: the export is already consumed. + pub abort_not_found: AtomicBool, + pub destroy_fails: AtomicBool, + pub spawn_fails: AtomicBool, + pub captures: AtomicUsize, + pub restores: AtomicUsize, + pub resumes: AtomicUsize, + pub aborts: AtomicUsize, + pub destroys: AtomicUsize, + pub spawns: AtomicUsize, + pub presetups: AtomicUsize, + pub live_refused: AtomicBool, + pub live_lost: AtomicBool, + pub peer_lost: AtomicBool, + pub block_capture: AtomicBool, + pub capture_entered: tokio::sync::Notify, +} +impl ScriptedHost { + fn new(clock: Arc, entropy: &dyn Entropy) -> Arc { + Arc::new(Self { + sandbox: SandboxId::from(entropy.uuid()), + snapshot: SnapshotId::from(entropy.uuid()), + clock, + capture_fails: AtomicBool::new(false), + restore_fails: AtomicBool::new(false), + resume_fails: AtomicBool::new(false), + resume_not_found: AtomicBool::new(false), + abort_not_found: AtomicBool::new(false), + destroy_fails: AtomicBool::new(false), + spawn_fails: AtomicBool::new(false), + captures: AtomicUsize::new(0), + restores: AtomicUsize::new(0), + resumes: AtomicUsize::new(0), + aborts: AtomicUsize::new(0), + destroys: AtomicUsize::new(0), + spawns: AtomicUsize::new(0), + presetups: AtomicUsize::new(0), + live_refused: AtomicBool::new(false), + live_lost: AtomicBool::new(false), + peer_lost: AtomicBool::new(false), + block_capture: AtomicBool::new(false), + capture_entered: tokio::sync::Notify::new(), + }) + } +} +#[async_trait] +impl HostClient for ScriptedHost { + async fn create(&self, _spec: SandboxSpec) -> Result { + Ok(self.sandbox) + } + async fn destroy( + &self, + _id: SandboxId, + _fence: engram_core::traits::SessionFence, + ) -> Result<(), SandboxError> { + self.destroys.fetch_add(1, Ordering::SeqCst); + if self.destroy_fails.load(Ordering::SeqCst) { + return Err(SandboxError::Timeout); + } + Ok(()) + } + async fn list(&self) -> Result, SandboxError> { + Ok(vec![]) + } + async fn probe_sandbox( + &self, + _id: SandboxId, + ) -> Result { + unimplemented!() + } + async fn exec_stream( + &self, + _id: SandboxId, + _cmd: ExecRequest, + ) -> Result { + unreachable!() + } + async fn snapshot_hold( + &self, + id: SandboxId, + fence: SessionFence, + ) -> Result { + self.pause(id, fence).await?; + self.snapshot(id, fence).await + } + + async fn snapshot( + &self, + _id: SandboxId, + _fence: engram_core::traits::SessionFence, + ) -> Result { + self.captures.fetch_add(1, Ordering::SeqCst); + if self.capture_fails.load(Ordering::SeqCst) { + return Err(SandboxError::Snapshot("capture failed".into())); + } + Ok(SnapshotMetadata { + id: self.snapshot, + size_bytes: 1024, + created_at: self.clock.now_utc(), + image_version: "test".into(), + disk_manifest: None, + memory_manifest: None, + base_memory_manifest: None, + migration_source: None, + source_sandbox_id: None, + state_blob_key: None, + sidecar_blob_key: None, + rootfs_blob_key: None, + working_set_blob_key: None, + aux_bundles: vec![], + paused_at: None, + peer_hints: Vec::new(), + }) + } + async fn commit_snapshot( + &self, + _id: SandboxId, + _fence: engram_core::traits::SessionFence, + ) -> Result<(), SandboxError> { + Ok(()) + } + async fn abort_snapshot( + &self, + _id: SandboxId, + _fence: engram_core::traits::SessionFence, + ) -> Result<(), SandboxError> { + Ok(()) + } + async fn restore( + &self, + _md: SnapshotMetadata, + _fence: engram_core::traits::SessionFence, + ) -> Result { + self.restores.fetch_add(1, Ordering::SeqCst); + if self.restore_fails.load(Ordering::SeqCst) { + return Err(SandboxError::Snapshot("restore failed".into())); + } + Ok(self.sandbox) + } + async fn start_agent( + &self, + _id: SandboxId, + _agent: engram_core::types::sandbox::AgentSpec, + _policy: engram_core::types::egress::SessionEgressPolicy, + _fence: engram_core::traits::SessionFence, + ) -> Result<(), SandboxError> { + self.spawns.fetch_add(1, Ordering::SeqCst); + if self.spawn_fails.load(Ordering::SeqCst) { + return Err(SandboxError::InvalidSpec("spawn rejected".into())); + } + Ok(()) + } + async fn guest_ip(&self, _id: SandboxId) -> Option { + None + } + async fn bind_session( + &self, + _session_id: SessionId, + _sandbox_id: SandboxId, + _binding_epoch: u64, + ) -> Result<(), engram_core::SandboxError> { + Ok(()) + } + async fn unbind_session(&self, _session_id: SessionId) {} + async fn send_prompt( + &self, + _sandbox_id: SandboxId, + _prompt_id: String, + _text: String, + _mode: Option, + ) -> Result<(), SandboxError> { + unreachable!() + } + async fn pause(&self, _id: SandboxId, _fence: SessionFence) -> Result<(), SandboxError> { + Ok(()) + } + async fn resume(&self, _id: SandboxId, _fence: SessionFence) -> Result<(), SandboxError> { + self.resumes.fetch_add(1, Ordering::SeqCst); + if self.resume_not_found.load(Ordering::SeqCst) { + Err(SandboxError::NotFound) + } else if self.resume_fails.load(Ordering::SeqCst) { + Err(SandboxError::Timeout) + } else { + Ok(()) + } + } + async fn migration_presetup( + &self, + _id: SandboxId, + _fence: SessionFence, + ) -> Result { + self.presetups.fetch_add(1, Ordering::SeqCst); + if self.live_refused.load(Ordering::SeqCst) { + return Err(SandboxError::InvalidSpec("swap enabled".into())); + } + let manifest = engram_core::types::manifest::ManifestRef { + manifest_id: self.snapshot.as_uuid(), + version: 1, + }; + Ok(engram_core::types::snapshot::MigrationPresetupOut { + export_id: "export-one".into(), + peer_token: "token-one".into(), + peer_port: 9000, + sidecar_json: vec![1], + memory_manifest_json: vec![2], + memory_manifest_ref: manifest, + disk_manifest_ref: Some(manifest), + hot_chunks: vec![[3; 32]], + }) + } + async fn migration_capture_postcopy( + &self, + _id: SandboxId, + export: &str, + _fence: SessionFence, + ) -> Result { + assert_eq!(export, "export-one"); + self.capture_entered.notify_one(); + if self.block_capture.load(Ordering::SeqCst) { + std::future::pending::<()>().await; + } + if self.live_lost.load(Ordering::SeqCst) { + return Err(SandboxError::NotFound); + } + Ok(engram_core::types::snapshot::PostCopyCaptureOut { + sealed_chunks: 1, + total_chunks: 1, + pause_ms: 0, + disk_drain_ms: 0, + vmstate_ms: 0, + scan_ms: 0, + sealed_disk_chunks: 1, + paused_at_unix_ms: 0, + }) + } + async fn migration_drain_wait( + &self, + _id: SandboxId, + ) -> Result { + if self.peer_lost.load(Ordering::SeqCst) { + Ok(engram_core::types::snapshot::DrainOutcome::PeerLost { + remaining: 1, + detail: "peer lost".into(), + }) + } else { + Ok(engram_core::types::snapshot::DrainOutcome::Done { + pulled: 1, + alt_sourced: 0, + zero_chunks: 0, + ms: 0, + }) + } + } + async fn migration_commit( + &self, + _id: SandboxId, + _export: &str, + _fence: SessionFence, + ) -> Result<(), SandboxError> { + Ok(()) + } + /// The live-move counterpart of `resume`: the same "source does not + /// answer" flag keeps a rollback pending until the abort is acknowledged. + async fn migration_abort( + &self, + _id: SandboxId, + _export: &str, + _fence: SessionFence, + ) -> Result<(), SandboxError> { + self.aborts.fetch_add(1, Ordering::SeqCst); + if self.abort_not_found.load(Ordering::SeqCst) { + Err(SandboxError::NotFound) + } else if self.resume_fails.load(Ordering::SeqCst) { + Err(SandboxError::Timeout) + } else { + Ok(()) + } + } + fn harness_dial(&self) -> HarnessDial { + HarnessDial::Vsock + } +} + +pub fn host_record(id: HostId, name: &str, now: chrono::DateTime) -> HostRecord { + HostRecord { + id, + hostname: name.to_string(), + cloud_metadata: Default::default(), + capacity: HostCapacity { + total_gb: 100, + used_gb: 0, + total_mib: 32_768, + used_mib: 0, + running_sandboxes: 0, + }, + utilization: Default::default(), + status: HostStatus::Ready, + last_heartbeat_at: now, + host_addr: Some("http://source:8080".into()), + ready_images: Vec::new(), + current_bundles: Vec::new(), + sandbox_bundles: Vec::new(), + cordoned: false, + cordon_owner: None, + cordon_reason: None, + retire_requested_at: None, + retired_at: None, + total_vcpus: 16, + wire_version: 1, + stages_images: false, + capabilities: Default::default(), + lease_expires_at: None, + lease_state: Default::default(), + lease_epoch: 0, + } +} + +pub struct Rig { + pub state: SharedState, + pub source: Arc, + pub dest: Arc, + pub row: TeleportRow, + pub op: SessionOp, + pub clock: Arc, + pub _dir: tempfile::TempDir, +} +impl Rig { + pub async fn new(meta: Arc, clock: Arc) -> Self { + let entropy: Arc = Arc::new(engram_sim::SimEntropy::seeded(77)); + let source = ScriptedHost::new(clock.clone(), entropy.as_ref()); + let dest = ScriptedHost::new(clock.clone(), entropy.as_ref()); + let source_id = HostId::from(entropy.uuid()); + let dest_id = HostId::from(entropy.uuid()); + meta.upsert_host(host_record(source_id, "source", clock.now_utc())) + .await + .unwrap(); + meta.upsert_host(host_record(dest_id, "dest", clock.now_utc())) + .await + .unwrap(); + let session = meta + .create_session(SessionSpec { + image: "test:teleport".into(), + mode: SessionMode::DevVm, + }) + .await + .unwrap(); + meta.assign_session_host(session, Some(source_id)) + .await + .unwrap(); + meta.transition_session_created(session, source.sandbox) + .await + .unwrap(); + meta.transition_session( + session, + SessionState::Active, + engram_core::types::BindingDisposition::Retain, + ) + .await + .unwrap(); + let EnqueueOutcome::Claimed(op) = meta + .op_enqueue_and_claim( + session, + OpKind::Teleport, + serde_json::json!({}), + None, + "test", + ) + .await + .unwrap() + else { + panic!("claimed") + }; + let TeleportAdmitOutcome::Admitted(row) = meta + .teleport_admit(TeleportAdmitRequest { + id: engram_core::TeleportId::from(entropy.uuid()), + session_id: session, + reason: TeleportReason::Ui, + epoch: op.epoch.unwrap(), + candidates: vec![dest_id], + pinned_dest: Some(dest_id), + mem_budget_mib: 128, + cpu_budget_vcpus: 1, + max_open_per_dest: 1, + live_capable: false, + }) + .await + .unwrap() + else { + panic!("admitted") + }; + let dir = tempfile::tempdir().unwrap(); + let registry = Arc::new(HostRegistry::new(meta.clone())); + registry.register(source_id, source.clone()); + registry.register(dest_id, dest.clone()); + registry.record_sandbox_owner(source.sandbox, source_id); + let blob = Arc::new(engram_storage_local::LocalBlobStorage::new( + dir.path().join("blobs"), + )); + let services = Services { + meta, + host: registry.clone(), + secrets: Arc::new(engram_secrets_dev::InMemorySecretStore::new()), + kek: Arc::new(engram_crypto::EnvVarKeyProvider::from_bytes( + [0; 32], "test", + )), + oci: Arc::new(engram_oci::OciClient::new(Arc::new( + engram_oci::AnonymousResolver, + ))), + auth_resolver: Arc::new(engram_oci::AnonymousResolver), + blob: blob.clone(), + chunk_store: engram_chunk_store::ChunkStore::new(blob), + host_pool: Arc::new(engram_protocol::grpc_pool::GrpcHostPool::new()), + materialize_dir: None, + clock: clock.clone(), + entropy, + }; + let state = Arc::new(AppState::new_with_registry( + CoordinatorConfig { + local_path: dir.path().to_path_buf(), + ..Default::default() + }, + services, + registry, + )); + Self { + state, + source, + dest, + row: *row, + op, + clock, + _dir: dir, + } + } + pub async fn sim() -> Self { + let clock = engram_sim::ManualClock::new(); + let meta = engram_sim::SimMetadataStore::new( + clock.clone(), + Arc::new(engram_sim::SimEntropy::seeded(88)), + ); + Self::new(meta, clock).await + } + pub async fn drive(&self) -> coordinator::session_ops::OpOutcome { + coordinator::session_verbs::dispatch(&coordinator::session_ops::OpCtx { + state: &self.state, + op: &self.op, + epoch: self.op.epoch.unwrap(), + }) + .await + } + pub async fn make_live(&self) { + assert!(self + .state + .services + .meta + .teleport_advance( + self.row.id, + TeleportPhase::Admitted, + TeleportPhase::Admitted, + TeleportPatch { + kind: Some(TeleportKind::Live), + ..Default::default() + }, + self.op.epoch.unwrap() + ) + .await + .unwrap()); + } + pub async fn reclaim(&mut self) { + self.clock.advance(std::time::Duration::from_secs(1)); + self.op = self + .state + .services + .meta + .op_reclaim_stale(std::time::Duration::ZERO, "successor") + .await + .unwrap() + .into_iter() + .find(|op| op.session_id == self.row.session_id) + .expect("reclaimed"); + } + pub fn spawn_drive(&self) -> tokio::task::JoinHandle { + let state = self.state.clone(); + let op = self.op.clone(); + tokio::spawn(async move { + coordinator::session_verbs::dispatch(&coordinator::session_ops::OpCtx { + state: &state, + epoch: op.epoch.unwrap(), + op: &op, + }) + .await + }) + } + pub async fn phase(&self) -> Option { + self.state + .services + .meta + .open_teleport_for_session(self.row.session_id) + .await + .unwrap() + .map(|r| r.phase) + } +} diff --git a/crates/engram-coordinator/tests/support/teleport_scenarios.rs b/crates/engram-coordinator/tests/support/teleport_scenarios.rs new file mode 100644 index 000000000..f6e27ed50 --- /dev/null +++ b/crates/engram-coordinator/tests/support/teleport_scenarios.rs @@ -0,0 +1,712 @@ +//! The same scripted phase scenarios run with SimMetadataStore and Postgres. +#[path = "teleport.rs"] +pub(super) mod support; +use super::coordinator; +use coordinator::session_ops::OpOutcome; +use engram_core::types::teleport::{SourceRelease, TeleportPatch, TeleportPhase}; +use engram_core::types::SessionState; +use std::sync::{atomic::Ordering, Arc}; +use support::Rig; + +macro_rules! scenario { + ($name:ident) => { + mod $name { + use super::*; + #[tokio::test] + async fn sim() { + super::$name(Rig::sim().await).await; + } + #[tokio::test] + #[ignore = "requires live Postgres at ENGRAM_TEST_DATABASE_URL"] + async fn pg() { + let Some(db) = engram_testkit::pg::fresh_db().await else { + return; + }; + let clock = engram_sim::ManualClock::new(); + let meta = Arc::new(db.store.clone().with_clock(clock.clone())); + super::$name(Rig::new(meta, clock).await).await; + } + } + }; +} +async fn snapshot_teleport_end_to_end(rig: Rig) { + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.phase().await, None); + let session = rig + .state + .services + .meta + .get_session(rig.row.session_id) + .await + .unwrap(); + assert_eq!(session.status, SessionState::Active); + assert_eq!(session.host_id, Some(rig.row.dest_host_id)); + assert_eq!(session.sandbox_id, Some(rig.dest.sandbox)); + assert_eq!(rig.source.captures.load(Ordering::SeqCst), 1); + assert_eq!(rig.dest.restores.load(Ordering::SeqCst), 1); + assert_eq!(rig.source.destroys.load(Ordering::SeqCst), 1); + assert_eq!(rig.dest.spawns.load(Ordering::SeqCst), 0); + assert_eq!( + rig.state + .services + .meta + .sandbox_tombstones_for_host(rig.row.source_host_id) + .await + .unwrap(), + vec![rig.source.sandbox] + ); +} +scenario!(snapshot_teleport_end_to_end); + +async fn rollback_pending_until_source_resume_acks(rig: Rig) { + rig.source.capture_fails.store(true, Ordering::SeqCst); + rig.source.resume_fails.store(true, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::RetryAfter(_, _))); + assert_eq!(rig.phase().await, Some(TeleportPhase::RollingBack)); + assert_eq!( + rig.state + .services + .meta + .get_session(rig.row.session_id) + .await + .unwrap() + .status, + SessionState::Evacuating + ); + rig.source.resume_fails.store(false, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.phase().await, None); + assert_eq!( + rig.state + .services + .meta + .get_session(rig.row.session_id) + .await + .unwrap() + .status, + SessionState::Active + ); + assert_eq!(rig.source.resumes.load(Ordering::SeqCst), 2); +} +scenario!(rollback_pending_until_source_resume_acks); + +async fn source_dead_at_release_is_source_host_gone(rig: Rig) { + rig.source.destroy_fails.store(true, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::RetryAfter(_, _))); + assert_eq!(rig.phase().await, Some(TeleportPhase::Attached)); + assert!(!rig + .state + .services + .meta + .teleport_release_source( + rig.row.id, + rig.op.epoch.unwrap(), + SourceRelease::SourceHostGone + ) + .await + .unwrap()); + rig.state + .services + .meta + .mark_host_dead_if_lease_expired(rig.row.source_host_id) + .await + .unwrap(); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.phase().await, None); +} +scenario!(source_dead_at_release_is_source_host_gone); + +async fn dead_host_sweep_skips_machine_owned_sources(rig: Rig) { + assert!(rig + .state + .services + .meta + .mark_host_dead_if_lease_expired(rig.row.source_host_id) + .await + .unwrap() + .is_empty()); + let session = rig + .state + .services + .meta + .get_session(rig.row.session_id) + .await + .unwrap(); + assert_eq!(session.status, SessionState::Evacuating); + assert_eq!(session.sandbox_id, Some(rig.source.sandbox)); +} +scenario!(dead_host_sweep_skips_machine_owned_sources); + +async fn fenced_step_stops_silently(mut rig: Rig) { + rig.op.epoch = Some(rig.op.epoch.unwrap() + 1); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.source.captures.load(Ordering::SeqCst), 0); + assert_eq!(rig.phase().await, Some(TeleportPhase::Admitted)); +} +scenario!(fenced_step_stops_silently); + +async fn commit_conflict_rolls_back_dest(rig: Rig) { + let meta = &rig.state.services.meta; + let epoch = rig.op.epoch.unwrap(); + meta.teleport_advance( + rig.row.id, + TeleportPhase::Admitted, + TeleportPhase::Captured, + TeleportPatch::default(), + epoch, + ) + .await + .unwrap(); + meta.teleport_advance( + rig.row.id, + TeleportPhase::Captured, + TeleportPhase::Restored, + TeleportPatch { + dest_sandbox_id: Some(rig.dest.sandbox), + ..Default::default() + }, + epoch, + ) + .await + .unwrap(); + meta.assign_session_host(rig.row.session_id, Some(rig.row.dest_host_id)) + .await + .unwrap(); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.dest.destroys.load(Ordering::SeqCst), 1); + assert_eq!(rig.phase().await, None); +} +scenario!(commit_conflict_rolls_back_dest); + +async fn crash_at_every_phase_is_resumed_by_successor(mut rig: Rig, phase: TeleportPhase) { + use engram_core::traits::HostClient; + use engram_core::types::{snapshot::SnapshotRecord, BindingDisposition}; + let meta = &rig.state.services.meta; + let epoch = rig.op.epoch.unwrap(); + let fence = engram_core::traits::SessionFence { + session_id: rig.row.session_id, + epoch: epoch.try_into().unwrap(), + }; + if matches!( + phase, + TeleportPhase::Captured + | TeleportPhase::Restored + | TeleportPhase::Committed + | TeleportPhase::Attached + ) { + let snapshot = rig + .source + .snapshot_hold(rig.source.sandbox, fence) + .await + .unwrap(); + meta.fenced_record_snapshot( + SnapshotRecord { + id: snapshot.id, + session_id: Some(rig.row.session_id), + host_id: Some(rig.row.source_host_id), + image_version: snapshot.image_version.clone(), + size_bytes: snapshot.size_bytes, + created_at: snapshot.created_at, + last_accessed_at: snapshot.created_at, + disk_manifest: None, + memory_manifest: None, + recoverable: false, + aux_bundles: vec![], + events_cursor: None, + fc_snapshot_version: None, + }, + epoch, + ) + .await + .unwrap(); + meta.teleport_advance( + rig.row.id, + TeleportPhase::Admitted, + TeleportPhase::Captured, + TeleportPatch { + snapshot_id: Some(snapshot.id), + ..Default::default() + }, + epoch, + ) + .await + .unwrap(); + if phase != TeleportPhase::Captured { + let dest = rig.dest.restore(snapshot, fence).await.unwrap(); + meta.teleport_advance( + rig.row.id, + TeleportPhase::Captured, + TeleportPhase::Restored, + TeleportPatch { + dest_sandbox_id: Some(dest), + ..Default::default() + }, + epoch, + ) + .await + .unwrap(); + } + if matches!(phase, TeleportPhase::Committed | TeleportPhase::Attached) { + assert!(meta + .teleport_commit(rig.row.id, epoch) + .await + .unwrap() + .is_some()); + rig.state + .host_registry + .record_sandbox_owner(rig.dest.sandbox, rig.row.dest_host_id); + } + if phase == TeleportPhase::Attached { + meta.fenced_transition_session( + rig.row.session_id, + epoch, + SessionState::Active, + BindingDisposition::Retain, + ) + .await + .unwrap(); + meta.teleport_advance( + rig.row.id, + TeleportPhase::Committed, + TeleportPhase::Attached, + TeleportPatch::default(), + epoch, + ) + .await + .unwrap(); + } + } else if phase == TeleportPhase::RollingBack { + rig.source.capture_fails.store(true, Ordering::SeqCst); + rig.source.resume_fails.store(true, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::RetryAfter(_, _))); + rig.source.resume_fails.store(false, Ordering::SeqCst); + } + assert_eq!(rig.phase().await, Some(phase)); + rig.reclaim().await; + assert!( + matches!(rig.drive().await, OpOutcome::Done), + "phase {phase:?}" + ); + assert_eq!(rig.phase().await, None, "phase {phase:?}"); + let session = meta_session(&rig).await; + assert_eq!(session.status, SessionState::Active); + assert_eq!( + session.host_id, + Some(if phase == TeleportPhase::RollingBack { + rig.row.source_host_id + } else { + rig.row.dest_host_id + }) + ); + assert_eq!( + rig.source.captures.load(Ordering::SeqCst), + 1, + "no second capture after {phase:?}" + ); +} +async fn meta_session(rig: &Rig) -> engram_core::types::Session { + rig.state + .services + .meta + .get_session(rig.row.session_id) + .await + .unwrap() +} +mod crash_at_every_phase { + use super::*; + const PHASES: [TeleportPhase; 6] = [ + TeleportPhase::Admitted, + TeleportPhase::Captured, + TeleportPhase::Restored, + TeleportPhase::Committed, + TeleportPhase::Attached, + TeleportPhase::RollingBack, + ]; + #[tokio::test] + async fn sim() { + for phase in PHASES { + crash_at_every_phase_is_resumed_by_successor(Rig::sim().await, phase).await; + } + } + #[tokio::test] + #[ignore = "requires live Postgres at ENGRAM_TEST_DATABASE_URL"] + async fn pg() { + for phase in PHASES { + let Some(db) = engram_testkit::pg::fresh_db().await else { + return; + }; + let clock = engram_sim::ManualClock::new(); + let meta = Arc::new(db.store.clone().with_clock(clock.clone())); + crash_at_every_phase_is_resumed_by_successor(Rig::new(meta, clock).await, phase).await; + } + } +} + +async fn crash_after_presetup_reuses_live_payload(mut rig: Rig) { + rig.make_live().await; + rig.source.block_capture.store(true, Ordering::SeqCst); + let task = rig.spawn_drive(); + rig.source.capture_entered.notified().await; + let row = rig + .state + .services + .meta + .open_teleport_for_session(rig.row.session_id) + .await + .unwrap() + .unwrap(); + let payload = row.live_payload.unwrap(); + assert_eq!(row.export_id.as_deref(), Some("export-one")); + assert_eq!(payload["migration_source"]["peer_token"], "token-one"); + assert_eq!(payload["migration_source"]["peer_addr"], "source:9000"); + task.abort(); + assert!(task.await.unwrap_err().is_cancelled()); + rig.reclaim().await; + rig.source.block_capture.store(false, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.source.presetups.load(Ordering::SeqCst), 1); + assert_eq!(rig.phase().await, None); +} +scenario!(crash_after_presetup_reuses_live_payload); + +async fn live_refusal_downgrades_to_snapshot_in_row(rig: Rig) { + rig.make_live().await; + rig.source.live_refused.store(true, Ordering::SeqCst); + rig.source.destroy_fails.store(true, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::RetryAfter(_, _))); + let row = rig + .state + .services + .meta + .open_teleport_for_session(rig.row.session_id) + .await + .unwrap() + .unwrap(); + assert_eq!( + row.kind, + engram_core::types::teleport::TeleportKind::Snapshot + ); + assert_eq!(row.phase, TeleportPhase::Attached); + assert_eq!(rig.source.captures.load(Ordering::SeqCst), 1); +} +scenario!(live_refusal_downgrades_to_snapshot_in_row); + +async fn lost_live_export_rolls_back_with_durable_error(rig: Rig) { + rig.make_live().await; + rig.source.live_lost.store(true, Ordering::SeqCst); + rig.source.resume_fails.store(true, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::RetryAfter(_, _))); + let row = rig + .state + .services + .meta + .open_teleport_for_session(rig.row.session_id) + .await + .unwrap() + .unwrap(); + assert_eq!(row.phase, TeleportPhase::RollingBack); + assert_eq!(row.error.as_deref(), Some("live_export_lost")); + assert_eq!(rig.source.presetups.load(Ordering::SeqCst), 1); + // A live rollback releases the fenced export through migration_abort, + // never through a plain resume, and stays pending until the source + // acknowledges it. + assert_eq!(rig.source.aborts.load(Ordering::SeqCst), 1); + assert_eq!(rig.source.resumes.load(Ordering::SeqCst), 0); + rig.source.resume_fails.store(false, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.source.aborts.load(Ordering::SeqCst), 2); + assert_eq!(rig.source.resumes.load(Ordering::SeqCst), 0); + // A finished row is no longer open. + assert_eq!(rig.phase().await, None); + assert_eq!( + meta_session(&rig).await.status, + engram_core::types::SessionState::Active + ); +} +scenario!(lost_live_export_rolls_back_with_durable_error); + +/// A consumed export (an earlier abort passed its point of no return) is +/// not a resumed guest: the rollback still needs the resume ack before it +/// declares Active. +async fn consumed_live_export_rollback_waits_for_the_resume_ack(rig: Rig) { + rig.make_live().await; + rig.source.live_lost.store(true, Ordering::SeqCst); + rig.source.abort_not_found.store(true, Ordering::SeqCst); + rig.source.resume_fails.store(true, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::RetryAfter(_, _))); + assert_eq!(rig.phase().await, Some(TeleportPhase::RollingBack)); + assert_eq!(rig.source.aborts.load(Ordering::SeqCst), 1); + assert_eq!(rig.source.resumes.load(Ordering::SeqCst), 1); + assert_eq!( + meta_session(&rig).await.status, + engram_core::types::SessionState::Evacuating + ); + rig.source.resume_fails.store(false, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.source.resumes.load(Ordering::SeqCst), 2); + assert_eq!(rig.phase().await, None); + assert_eq!( + meta_session(&rig).await.status, + engram_core::types::SessionState::Active + ); +} +scenario!(consumed_live_export_rollback_waits_for_the_resume_ack); + +/// A source sandbox that no longer exists cannot be returned to: the +/// rollback settles the session honestly instead of retrying forever. +async fn rollback_with_a_destroyed_source_fails_the_move(rig: Rig) { + rig.source.capture_fails.store(true, Ordering::SeqCst); + rig.source.resume_not_found.store(true, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.phase().await, None); + let events = rig + .state + .services + .meta + .list_session_events_since(rig.row.session_id, -1, 100) + .await + .unwrap(); + let terminal: Vec<_> = events + .iter() + .filter(|e| e.kind == "teleport_finished") + .collect(); + assert_eq!(terminal.len(), 1); + assert_eq!(terminal[0].payload["outcome"], "failed"); + assert_eq!(terminal[0].payload["error"], "source_lost_during_rollback"); + // No durable snapshot in this rig: the session is Dead, not Idle. + assert_eq!( + meta_session(&rig).await.status, + engram_core::types::SessionState::Dead + ); +} +scenario!(rollback_with_a_destroyed_source_fails_the_move); + +/// A disk-only session (live disk manifest, no memory snapshot) is still +/// recoverable through the cold boot: a lost source settles it Idle, not +/// Dead, and keeps the manifest. +async fn rollback_with_a_destroyed_source_keeps_a_disk_only_session_idle(rig: Rig) { + let manifest = engram_core::types::manifest::ManifestRef::new(); + rig.state + .services + .meta + .update_live_disk_manifest(rig.row.session_id, rig.row.source_sandbox_id, manifest) + .await + .unwrap(); + rig.source.capture_fails.store(true, Ordering::SeqCst); + rig.source.resume_not_found.store(true, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert_eq!(rig.phase().await, None); + let session = meta_session(&rig).await; + assert_eq!(session.status, engram_core::types::SessionState::Idle); + assert_eq!(session.sandbox_id, None); + assert_eq!(session.live_disk_manifest, Some(manifest)); +} +scenario!(rollback_with_a_destroyed_source_keeps_a_disk_only_session_idle); + +async fn another_admission(rig: &Rig) -> engram_core::types::teleport::TeleportAdmitRequest { + use engram_core::types::teleport::*; + let meta = &rig.state.services.meta; + let sid = meta + .create_session(engram_core::types::SessionSpec { + image: "test:teleport".into(), + mode: engram_core::types::SessionMode::DevVm, + }) + .await + .unwrap(); + meta.assign_session_host(sid, Some(rig.row.source_host_id)) + .await + .unwrap(); + meta.transition_session_created( + sid, + engram_core::SandboxId::from(rig.state.services.entropy.uuid()), + ) + .await + .unwrap(); + meta.transition_session( + sid, + SessionState::Active, + engram_core::types::BindingDisposition::Retain, + ) + .await + .unwrap(); + TeleportAdmitRequest { + id: engram_core::TeleportId::from(rig.state.services.entropy.uuid()), + session_id: sid, + reason: TeleportReason::Ui, + epoch: 0, + candidates: vec![rig.row.dest_host_id], + pinned_dest: Some(rig.row.dest_host_id), + mem_budget_mib: 128, + cpu_budget_vcpus: 1, + max_open_per_dest: 1, + live_capable: false, + } +} +async fn admit_no_fit_leaves_session_active_and_no_row(rig: Rig) { + let req = another_admission(&rig).await; + let sid = req.session_id; + assert!(matches!( + rig.state.services.meta.teleport_admit(req).await.unwrap(), + engram_core::types::teleport::TeleportAdmitOutcome::NoFit + )); + assert_eq!( + rig.state + .services + .meta + .get_session(sid) + .await + .unwrap() + .status, + SessionState::Active + ); + assert!(rig + .state + .services + .meta + .open_teleport_for_session(sid) + .await + .unwrap() + .is_none()); +} +scenario!(admit_no_fit_leaves_session_active_and_no_row); + +async fn two_teleports_to_one_dest_respect_max_open(rig: Rig) { + let mut left = another_admission(&rig).await; + let mut right = another_admission(&rig).await; + left.max_open_per_dest = 2; + right.max_open_per_dest = 2; + let meta = &rig.state.services.meta; + let (a, b) = tokio::join!(meta.teleport_admit(left), meta.teleport_admit(right)); + use engram_core::types::teleport::TeleportAdmitOutcome; + assert_eq!( + [a.unwrap(), b.unwrap()] + .iter() + .filter(|r| matches!(r, TeleportAdmitOutcome::Admitted(_))) + .count(), + 1 + ); + assert_eq!(meta.list_open_teleports().await.unwrap().len(), 2); +} +scenario!(two_teleports_to_one_dest_respect_max_open); + +async fn retire_host_grants_only_after_empty_heartbeat_then_delete_host_ok(rig: Rig) { + use engram_core::types::host::{ + CordonOwner, DeleteHostOutcome, HostHeartbeat, RetirementGrant, + }; + let meta = &rig.state.services.meta; + let source = rig.row.source_host_id; + meta.request_host_retirement( + source, + CordonOwner::Admin, + "test", + rig.state.services.clock.now_utc(), + ) + .await + .unwrap(); + assert!(matches!( + meta.grant_host_retirement(source, rig.state.services.clock.now_utc()) + .await + .unwrap(), + RetirementGrant::Blocked(_) + )); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + assert!(matches!( + meta.grant_host_retirement(source, rig.state.services.clock.now_utc()) + .await + .unwrap(), + RetirementGrant::Blocked(_) + )); + rig.clock.advance(std::time::Duration::from_secs(1)); + let h = meta.get_host(source).await.unwrap().unwrap(); + meta.touch_host_heartbeat( + source, + HostHeartbeat { + status: h.status, + capacity: h.capacity, + utilization: h.utilization, + ready_images: vec![], + current_bundles: vec![], + sandbox_bundles: vec![], + total_vcpus: h.total_vcpus, + wire_version: h.wire_version, + stages_images: false, + capabilities: h.capabilities, + lease_renew_until: None, + }, + ) + .await + .unwrap(); + assert!( + matches!( + meta.grant_host_retirement(source, rig.state.services.clock.now_utc()) + .await + .unwrap(), + RetirementGrant::Blocked(_) + ), + "tombstone still blocks grant" + ); + meta.ack_sandbox_tombstones_by_absence(source, &[]) + .await + .unwrap(); + coordinator::teleport::run_once(&Default::default(), &rig.state) + .await + .unwrap(); + assert_eq!( + meta.get_host(source).await.unwrap().unwrap().status, + engram_core::types::HostStatus::Retired + ); + assert_eq!( + meta.delete_host(source).await.unwrap(), + DeleteHostOutcome::Deleted + ); + assert_eq!( + meta.delete_host(source).await.unwrap(), + DeleteHostOutcome::Deleted + ); +} +scenario!(retire_host_grants_only_after_empty_heartbeat_then_delete_host_ok); + +async fn live_peer_lost_fails_to_idle_with_durable_row(rig: Rig) { + rig.make_live().await; + let meta = &rig.state.services.meta; + let now = rig.state.services.clock.now_utc(); + meta.fenced_record_snapshot( + engram_core::types::snapshot::SnapshotRecord { + id: rig.source.snapshot, + session_id: Some(rig.row.session_id), + host_id: Some(rig.row.source_host_id), + image_version: "test".into(), + size_bytes: 1, + created_at: now, + last_accessed_at: now, + disk_manifest: None, + memory_manifest: None, + recoverable: true, + aux_bundles: vec![], + events_cursor: None, + fc_snapshot_version: None, + }, + rig.op.epoch.unwrap(), + ) + .await + .unwrap(); + rig.dest.peer_lost.store(true, Ordering::SeqCst); + assert!(matches!(rig.drive().await, OpOutcome::Done)); + let s = meta.get_session(rig.row.session_id).await.unwrap(); + assert_eq!(s.status, SessionState::Idle); + assert_eq!(s.sandbox_id, None); + assert_eq!(rig.phase().await, None); + assert_eq!(rig.dest.destroys.load(Ordering::SeqCst), 1); + let events = meta + .list_session_events_since(rig.row.session_id, -1, 100) + .await + .unwrap(); + let terminal: Vec<_> = events + .iter() + .filter(|e| e.kind == "teleport_finished") + .collect(); + assert_eq!(terminal.len(), 1); + assert_eq!(terminal[0].payload["outcome"], "failed"); + assert_eq!(terminal[0].payload["error"], "peer_lost"); +} +scenario!(live_peer_lost_fails_to_idle_with_durable_row); diff --git a/crates/engram-coordinator/tests/teleport_live_pg.rs b/crates/engram-coordinator/tests/teleport_live_pg.rs new file mode 100644 index 000000000..d4ea2c000 --- /dev/null +++ b/crates/engram-coordinator/tests/teleport_live_pg.rs @@ -0,0 +1,3 @@ +use engram_coordinator as coordinator; +#[path = "support/teleport_scenarios.rs"] +mod scenarios; diff --git a/crates/engram-core/src/traits/host_client.rs b/crates/engram-core/src/traits/host_client.rs index d7784e694..799d459f6 100644 --- a/crates/engram-core/src/traits/host_client.rs +++ b/crates/engram-core/src/traits/host_client.rs @@ -172,6 +172,17 @@ pub trait HostClient: Send + Sync { )) } + /// Capture a portable snapshot and keep the source paused until resume or destroy. + async fn snapshot_hold( + &self, + _id: SandboxId, + _fence: SessionFence, + ) -> Result { + Err(SandboxError::InvalidSpec( + "snapshot hold is unsupported".into(), + )) + } + async fn snapshot( &self, id: SandboxId, diff --git a/crates/engram-core/src/traits/metadata.rs b/crates/engram-core/src/traits/metadata.rs index e5e3cd445..1780e8ba3 100644 --- a/crates/engram-core/src/traits/metadata.rs +++ b/crates/engram-core/src/traits/metadata.rs @@ -1,4 +1,9 @@ use crate::types::host::DeleteHostOutcome; +use crate::types::ids::TeleportId; +use crate::types::teleport::{ + SourceRelease, TeleportAdmitOutcome, TeleportAdmitRequest, TeleportPatch, TeleportPhase, + TeleportRow, TeleportSettle, +}; use async_trait::async_trait; use crate::error::MetaError; @@ -158,6 +163,70 @@ impl ExecOutputStream { /// we can support SQLite for embedded deployments later. #[async_trait] pub trait MetadataStore: Send + Sync { + /// Current binding generation and highest attached generation. + async fn session_binding_generations(&self, _id: SessionId) -> Result<(u64, u64), MetaError> { + unimplemented!("session_binding_generations") + } + + /// Admit a move and reserve its destination in one transaction. + async fn teleport_admit( + &self, + _req: TeleportAdmitRequest, + ) -> Result { + unimplemented!("teleport_admit") + } + /// Compare the phase and session fence before applying the patch. + async fn teleport_advance( + &self, + _id: TeleportId, + _from: TeleportPhase, + _to: TeleportPhase, + _patch: TeleportPatch, + _epoch: i64, + ) -> Result { + unimplemented!("teleport_advance") + } + /// Move the binding and phase together; mint one binding generation. + async fn teleport_commit( + &self, + _id: TeleportId, + _epoch: i64, + ) -> Result, MetaError> { + unimplemented!("teleport_commit") + } + /// Finish only after source teardown has been confirmed. + async fn teleport_release_source( + &self, + _id: TeleportId, + _epoch: i64, + _how: SourceRelease, + ) -> Result { + unimplemented!("teleport_release_source") + } + /// Mark an open row `failed` and settle the session in one fenced + /// transaction (see [`TeleportSettle`]). `false` when the row is already + /// terminal or the epoch moved; `Conflict` on an illegal session edge. + async fn teleport_settle( + &self, + _id: TeleportId, + _epoch: i64, + _settle: TeleportSettle, + ) -> Result { + unimplemented!("teleport_settle") + } + async fn teleport_abort(&self, _id: TeleportId, _epoch: i64) -> Result { + unimplemented!("teleport_abort") + } + async fn list_open_teleports(&self) -> Result, MetaError> { + unimplemented!("list_open_teleports") + } + async fn open_teleport_for_session( + &self, + _sid: SessionId, + ) -> Result, MetaError> { + unimplemented!("open_teleport_for_session") + } + // ---- liveness ---- // // Cheap connectivity check for readiness probes. Default is @@ -383,33 +452,6 @@ pub trait MetadataStore: Send + Sync { Ok(QueuedDemand::default()) } - /// ADR 0047 (was `state.teleport_targets`): pin / clear the - /// operator-chosen teleport destination on the session row. The - /// evac scanner — on ANY replica — honors the pin as its required - /// placement. Setting a pin stamps `teleport_target_set_at = NOW()` - /// (issue #214) so the scanner can age out a stale leaked pin; - /// clearing (`None`) clears the stamp too. Default impls (mocks): - /// no-op / no pin. - async fn set_teleport_target( - &self, - _id: SessionId, - _target: Option, - ) -> Result<(), MetaError> { - Ok(()) - } - /// Read the operator-pinned teleport destination and the instant it - /// was set, if any. Issue #214: the `set_at` lets the evac scanner - /// ignore + clear a pin older than a TTL — degrading any future pin - /// leak to default placement instead of a strict hijack. A `None` - /// timestamp (pin set before the 0065 migration) is treated as - /// not-aged by the scanner. Default impls (mocks): no pin. - async fn get_teleport_target( - &self, - _id: SessionId, - ) -> Result>)>, MetaError> { - Ok(None) - } - /// ADR 0047 (was `state.git_broker_tokens`): the KEK-sealed /// per-session broker token. `insert_broker_token` is /// first-writer-wins (`ON CONFLICT DO NOTHING`) and returns whether @@ -3714,26 +3756,6 @@ pub trait MetadataStore: Send + Sync { Ok(()) } - // ---------------------------------------------------------------- - // ADR 0018 commit 12b — evac_resumer scanner support. - // - // The scanner polls `Evacuating` sessions, picks a peer host, - // and drives `Evacuating → Created → Active`. The retry counter - // is a side-car on the `sessions` row (column `evac_attempts`, - // migration 0037). The PG-backed `transition_session(Evacuating)` - // resets the counter to 0 in the same UPDATE so re-entry from a - // fresh drain starts fresh; bumps happen via `bump_evac_attempts` - // (atomic UPDATE ... RETURNING). - // ---------------------------------------------------------------- - - /// Sessions currently in `Evacuating`, paired with their current - /// `evac_attempts` count. The scanner uses this on every tick. - /// Default `Ok(vec![])` keeps in-memory mocks quiet; PG impl - /// runs an indexed `WHERE status = 'evacuating'` query. - async fn list_evacuating_sessions(&self) -> Result, MetaError> { - Ok(Vec::new()) - } - /// Input to the dead-host driver's straggler sweep: `HostLost` rows /// whose inline stage-2 transition never ran or failed. These arise /// when eviction exhausts its retry budget or a coordinator replica @@ -3744,15 +3766,6 @@ pub trait MetadataStore: Send + Sync { Ok(Vec::new()) } - /// Atomically `evac_attempts = evac_attempts + 1 RETURNING - /// evac_attempts`. Scanner calls this before each resume attempt; - /// when the returned count exceeds the budget, scanner gives up - /// and falls back to Idle. Default returns 1 so test mocks can - /// observe the bump without persisting state. - async fn bump_evac_attempts(&self, _session_id: SessionId) -> Result { - Ok(1) - } - // ---------------------------------------------------------------- // ADR 0034 — eviction scanner + idle-detection backstop support. // diff --git a/crates/engram-core/src/traits/sandbox.rs b/crates/engram-core/src/traits/sandbox.rs index bd6af460c..181f25b0f 100644 --- a/crates/engram-core/src/traits/sandbox.rs +++ b/crates/engram-core/src/traits/sandbox.rs @@ -355,6 +355,18 @@ pub trait SandboxBackend: Send + Sync { /// coord persists to the `snapshots` row. async fn snapshot(&self, id: SandboxId) -> Result; + /// Capture without resuming. `diff` requests the same sparse format as snapshot_diff. + /// A caller owns resume or destroy on every exit, including cancellation. + async fn snapshot_hold( + &self, + _id: SandboxId, + _diff: bool, + ) -> Result { + Err(SandboxError::InvalidSpec( + "snapshot hold is unsupported".into(), + )) + } + /// ADR 0045 D5: the pause-side half of an eviction snapshot — see /// `HostClient::snapshot_begin`. Backends that can't background the /// upload keep the default (callers fall back to [`Self::snapshot`]). diff --git a/crates/engram-core/src/types/evacuation.rs b/crates/engram-core/src/types/evacuation.rs deleted file mode 100644 index aa5abf42e..000000000 --- a/crates/engram-core/src/types/evacuation.rs +++ /dev/null @@ -1,115 +0,0 @@ -//! ADR 0018: types for the session-evacuation primitive. -//! -//! Evacuation is the "move a sandbox from its current host to a peer" -//! operation that turns `Sandbox` into a host-fungible value per ADR -//! 0015 §M4. The `sandbox_id` token in the registry / DB is -//! substitutable across an evac — new id, same logical session. -//! -//! Lives in `engram-core::types` because both the trait surface -//! (`HostClient::evacuate`) and the orchestrator (`HostRegistry`) need -//! the receipt + loss shape, and they live in different crates. - -use serde::{Deserialize, Serialize}; - -use super::ids::{HostId, SandboxId}; - -/// Outcome of a successful evacuation. The caller usually only cares -/// about `new_sandbox_id` (cache invalidation) and `loss` (telemetry). -/// `new_host_id` is exposed so admin endpoints can echo the placement -/// decision back to the operator. -/// -/// `session_id` is intentionally NOT part of the receipt: the -/// `session_id` is the stable handle across an evac per ADR 0015 §M4 -/// ("the `sandbox_id` token can be substituted, but the same logical -/// Sandbox from the session's perspective"). The caller already knows -/// it. -#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] -pub struct EvacReceipt { - pub binding_epoch: u64, - pub new_host_id: HostId, - pub new_sandbox_id: SandboxId, - pub loss: EvacLoss, -} - -/// What, if anything, was sacrificed by the evacuation. The graceful- -/// drain path (alive source) is always `None`; the dead-source paths -/// degrade as needed per ADR 0016 §"What gets easier". -/// -/// The `reason` field on `Memory` is conventionally one of the short -/// kebab-case tags the orchestrator emits ("source-dead-no-snapshot", -/// "source-disk-only-available", etc.). Operators key alerts on it; new -/// reasons added by future loss paths should be documented at the call -/// site that produces them. -#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] -#[serde(tag = "kind", rename_all = "snake_case")] -pub enum EvacLoss { - /// Memory + disk both preserved. Alive-source drain path. - None, - /// Memory state was not recoverable; disk was restored from - /// `sessions.live_disk_manifest_*` (ADR 0016 Phase B) or from the - /// most recent recoverable snapshot's disk manifest. - Memory { reason: String }, -} - -impl EvacLoss { - /// Stable tag used in metrics / log fields. Avoids the verbose - /// JSON form when only the discriminant matters. - pub const fn as_str(&self) -> &'static str { - match self { - Self::None => "none", - Self::Memory { .. } => "memory", - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::types::ids::{HostId, SandboxId}; - - #[test] - fn evac_loss_as_str_matches_discriminant() { - assert_eq!(EvacLoss::None.as_str(), "none"); - assert_eq!( - EvacLoss::Memory { - reason: "source-dead".into() - } - .as_str(), - "memory" - ); - } - - #[test] - fn evac_loss_round_trips_through_json() { - // The wire shape is the load-bearing contract — admin endpoint - // responses and telemetry both depend on the `kind` tag layout. - let none = serde_json::to_value(EvacLoss::None).unwrap(); - assert_eq!(none, serde_json::json!({"kind": "none"})); - let mem = serde_json::to_value(EvacLoss::Memory { - reason: "source-dead-no-snapshot".into(), - }) - .unwrap(); - assert_eq!( - mem, - serde_json::json!({"kind": "memory", "reason": "source-dead-no-snapshot"}) - ); - - let back: EvacLoss = serde_json::from_value(serde_json::json!({"kind": "none"})).unwrap(); - assert_eq!(back, EvacLoss::None); - } - - #[test] - fn evac_receipt_round_trips_through_json() { - let receipt = EvacReceipt { - binding_epoch: 1, - new_host_id: HostId::new(), - new_sandbox_id: SandboxId::new(), - loss: EvacLoss::Memory { - reason: "source-dead".into(), - }, - }; - let blob = serde_json::to_string(&receipt).unwrap(); - let back: EvacReceipt = serde_json::from_str(&blob).unwrap(); - assert_eq!(back, receipt); - } -} diff --git a/crates/engram-core/src/types/mod.rs b/crates/engram-core/src/types/mod.rs index 583c1ce93..89a60afc0 100644 --- a/crates/engram-core/src/types/mod.rs +++ b/crates/engram-core/src/types/mod.rs @@ -9,7 +9,6 @@ pub mod connector_oauth; pub mod cow_state; pub mod egress; pub mod endpoints; -pub mod evacuation; pub mod event; pub mod harness; pub mod host; @@ -37,7 +36,6 @@ pub use catalog::*; pub use cow_state::*; pub use egress::*; pub use endpoints::*; -pub use evacuation::*; pub use event::*; pub use harness::*; pub use host::*; diff --git a/crates/engram-core/src/types/session.rs b/crates/engram-core/src/types/session.rs index 2a4906822..d1aff15db 100644 --- a/crates/engram-core/src/types/session.rs +++ b/crates/engram-core/src/types/session.rs @@ -84,25 +84,9 @@ pub enum SessionState { /// there is nothing to recover. The detector no longer routes /// into `Evacuating` (the reactive auto-evac is retired). HostLost, - /// ADR 0018 commit 12: session is mid-relocation. The source host - /// has paused FC, flushed dirty pages, captured a memory snapshot, - /// destroyed the local sandbox, and durably committed both memory - /// and disk manifests to BlobStorage. The coord-side - /// `evac_resumer` background task scans for sessions in this - /// state and drives `Evacuating → Created → Active` on a peer - /// host via the same `resume_session` machinery `/resume from - /// Idle` uses. After 20 failed peer-pick / restore attempts - /// (~3 min), falls back to `Idle` so a user `/resume` can drive - /// it forward by hand. Reached **only** from `Active` via operator - /// drain (`POST /api/admin/sessions/:id/evacuate` or - /// `POST /api/admin/hosts/:id/drain`) — ADR 0044 K3. As of ADR - /// 0045 Phase A the dead-host detector no longer routes here (the - /// reactive auto-evac is retired), and `HostLost → Evacuating` is - /// no longer a legal edge. Post-#896 the source binding is - /// RETAINED through `Evacuating` (ADR 0090: a destroy - /// acknowledgement is not ownership proof) — the evac resumer - /// clears it under its claim only after the source teardown is - /// positively confirmed. + /// ADR 0123: a durable teleport owns this relocation. The source binding + /// stays in place until commit atomically installs the destination binding. + /// The source remains paused until rollback resumes it or release destroys it. Evacuating, /// ADR 0034: durable idle-eviction intent marker. The candidates /// handler (or the PG detection backstop) transitions diff --git a/crates/engram-core/src/types/teleport.rs b/crates/engram-core/src/types/teleport.rs index 97436d4d4..648904654 100644 --- a/crates/engram-core/src/types/teleport.rs +++ b/crates/engram-core/src/types/teleport.rs @@ -1,4 +1,5 @@ use super::ids::{HostId, SandboxId, SessionId, SnapshotId, TeleportId}; +use super::session::SessionState; use chrono::{DateTime, Utc}; use serde::{Deserialize, Serialize}; #[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize, Deserialize)] @@ -73,10 +74,24 @@ impl TeleportPhase { .collect::>() .join(",") } + /// Phases in which the SOURCE still holds a frozen VM or a live export + /// after the session's own reservation moved to the destination at + /// commit. The source budget stays reserved until release so a + /// placement cannot reuse RAM the paused VM still occupies. + pub const fn source_reserving_phases() -> &'static [&'static str] { + &["committed", "attached"] + } + pub fn source_reserving_phases_sql() -> String { + Self::source_reserving_phases() + .iter() + .map(|s| format!("'{s}'")) + .collect::>() + .join(",") + } pub const fn can_transition_to(self, target: Self) -> bool { use TeleportPhase::*; match self { - Admitted => matches!(target, Captured | RollingBack), + Admitted => matches!(target, Admitted | Captured | RollingBack), Captured => matches!(target, Restored | RollingBack), Restored => matches!(target, Committed | RollingBack), Committed => matches!(target, Attached | Failed), @@ -102,6 +117,7 @@ pub struct TeleportRow { pub cpu_budget_vcpus: i32, pub snapshot_id: Option, pub export_id: Option, + pub live_payload: Option, pub attempts: i32, pub error: Option, pub created_at: DateTime, @@ -130,7 +146,8 @@ mod tests { for to in ALL { let expected = matches!( (from, to), - (Admitted, Captured) + (Admitted, Admitted) + | (Admitted, Captured) | (Admitted, RollingBack) | (Captured, Restored) | (Captured, RollingBack) @@ -166,3 +183,60 @@ mod tests { } } } + +#[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum TeleportOutcome { + Done, + Aborted, + Failed, +} + +/// Admission holds the session and destination placement locks until commit. +#[derive(Clone, Debug)] +pub struct TeleportAdmitRequest { + pub id: TeleportId, + pub session_id: SessionId, + pub reason: TeleportReason, + pub epoch: i64, + pub candidates: Vec, + pub pinned_dest: Option, + pub mem_budget_mib: i64, + pub cpu_budget_vcpus: i64, + pub max_open_per_dest: u32, + pub live_capable: bool, +} +#[derive(Clone, Debug)] +pub enum TeleportAdmitOutcome { + Admitted(Box), + NoFit, + SessionNotActive(SessionState), + Fenced, +} +#[derive(Clone, Debug, Default)] +pub struct TeleportPatch { + pub dest_sandbox_id: Option, + pub snapshot_id: Option, + pub export_id: Option, + pub live_payload: Option, + pub kind: Option, + pub error: Option, +} +/// One fenced settlement of a failed move. The session's terminal state +/// (always detaching the binding), the source tombstone, the row's `failed` +/// phase, and both events land in ONE transaction under the session's +/// `current_epoch`, so a stale driver can never write a tombstone or a +/// terminal row after a successor took the lane. +#[derive(Clone, Debug, Default)] +pub struct TeleportSettle { + pub error: String, + /// `Some(target)` settles the session; `None` leaves it as it is (a + /// crashed predecessor or the dead-host sweep already settled it). + pub session: Option, + pub entomb_source: bool, +} +#[derive(Clone, Copy, Debug)] +pub enum SourceRelease { + DestroyAcked, + SourceHostGone, +} diff --git a/crates/engram-dst-cosim/src/bridge.rs b/crates/engram-dst-cosim/src/bridge.rs index b3cb97420..1632bf43e 100644 --- a/crates/engram-dst-cosim/src/bridge.rs +++ b/crates/engram-dst-cosim/src/bridge.rs @@ -146,6 +146,15 @@ impl HostClient for CosimHostClient { // result. The checkpoint-severance composition needs only attach/tail; // cancel's real guest verb is covered at the agentd boundary. + async fn snapshot_hold( + &self, + id: SandboxId, + fence: SessionFence, + ) -> Result { + self.pause(id, fence).await?; + self.snapshot(id, fence).await + } + async fn snapshot( &self, id: SandboxId, diff --git a/crates/engram-dst/src/invariants.rs b/crates/engram-dst/src/invariants.rs index 24e281e2e..6bb6198ae 100644 --- a/crates/engram-dst/src/invariants.rs +++ b/crates/engram-dst/src/invariants.rs @@ -344,6 +344,13 @@ fn placement_accounting(world: &SimWorld) -> Result<(), Violation> { } } } + for row in db.teleports.values() { + if engram_core::types::teleport::TeleportPhase::dest_reserving_phases() + .contains(&row.phase.as_str()) + { + *reserved.entry(row.dest_host_id).or_default() += row.mem_budget_mib; + } + } for (host, mem) in reserved { if let Some(h) = db.hosts.get(&host) { let alloc = h.utilization.allocatable_mib as i64; @@ -429,6 +436,27 @@ fn sandbox_owners_agree(world: &SimWorld) -> Result<(), Violation> { fn snapshot_safety(world: &SimWorld) -> Result<(), Violation> { world.meta.with_db(|db| { for row in db.sessions.values() { + // Admission precedes capture. A live move can also retain its only + // copy on the source until drain completes; its row owns recovery. + let teleport_owns_copy = row.session.status == SessionState::Evacuating + && db.teleports.values().any(|t| { + t.session_id == row.session.id + && !t.phase.is_terminal() + && (t.kind == engram_core::types::teleport::TeleportKind::Live + || matches!( + t.phase, + engram_core::types::teleport::TeleportPhase::Admitted + | engram_core::types::teleport::TeleportPhase::RollingBack + )) + && ((row.session.host_id == Some(t.source_host_id) + && row.session.sandbox_id == Some(t.source_sandbox_id)) + || (row.session.host_id == Some(t.dest_host_id) + && row.session.sandbox_id == t.dest_sandbox_id + && t.dest_sandbox_id.is_some())) + }); + if teleport_owns_copy { + continue; + } let requires_durable = matches!( row.session.status, SessionState::Idle | SessionState::Evacuating @@ -522,6 +550,30 @@ fn bound_sessions_point_at_live_hosts(world: &SimWorld) -> Result<(), Violation> /// convergence, no session may be stuck in a transient state and no /// op may be left undriven. pub fn check_quiescence(world: &SimWorld) -> Result<(), Violation> { + world.meta.with_db(|db| { + if let Some(row) = db.teleports.values().find(|r| !r.phase.is_terminal()) { + return Err(Violation { + invariant: "quiescence-open-teleport", + detail: format!("teleport {} remains {:?}", row.id, row.phase), + }); + } + for host in db + .hosts + .values() + .filter(|h| h.status == engram_core::types::HostStatus::Retired) + { + if db.sessions.values().any(|s| { + s.session.host_id == Some(host.id) && s.session.status.reserves_host_memory() + }) || db.sandbox_tombstones.keys().any(|(h, _)| *h == host.id) + { + return Err(Violation { + invariant: "quiescence-retired-host-owned", + detail: format!("retired host {} retains work", host.id), + }); + } + } + Ok(()) + })?; no_op_dropped(world)?; no_orphan_sandboxes(world)?; ttft_liveness(world)?; diff --git a/crates/engram-dst/src/scheduler.rs b/crates/engram-dst/src/scheduler.rs index 0cb98b48c..e8d59e630 100644 --- a/crates/engram-dst/src/scheduler.rs +++ b/crates/engram-dst/src/scheduler.rs @@ -47,7 +47,7 @@ pub enum DriverKind { OpReclaim, IdleDetector, IdleEvictor, - EvacResumer, + Teleport, /// Empty-sweep boundary: the sim world creates no enable jobs, so this /// exercises deadline/list/claim calls against real SimMeta methods. /// Deeper capture legs remain panic-stubbed per ADR 0098's deviation. @@ -74,7 +74,7 @@ const DRIVERS: [DriverKind; 13] = [ DriverKind::OpReclaim, DriverKind::IdleDetector, DriverKind::IdleEvictor, - DriverKind::EvacResumer, + DriverKind::Teleport, DriverKind::EnableScanner, DriverKind::CheckpointRetention, DriverKind::BaseSnapshotRetention, @@ -153,22 +153,10 @@ pub enum Step { /// An explicit `delete_session` through the real Destroy op + teardown. /// A destroy that acks feeds the acked-destroy-never-resurrects oracle. Destroy, - /// An operator draining a host: cordon + evacuate its bound sessions to - /// Evacuating (the EvacResumer driver then re-homes them — the #775 - /// dormant leg). FOLDED into both profile menus at small weight (#800), - /// completing the dormant-Evacuating-leg coverage. Both wave-4 blockers - /// are now cleared: (1) #799's in-memory `MemBlobStorage` removed the - /// evict-pipeline fs race (determinism-audit item 7), and (2) #800's - /// RESERVED evac placement closed the capacity-soft over-reservation - /// (evac now QUEUES rather than binding a measured-full survivor). The - /// full gRPC handler (JoinSet, live-teleport preview, don't-strand - /// guard) still lives only in the dedicated tests (tests/api_surface.rs - /// drives the real `admin_drain_host`; tests/workload_verbs.rs the - /// sequential cordon+evict); the swarm arm drives the same - /// cordon+evict-to-Evacuating pipeline sequentially. The host index is - /// drawn from WORLD entropy (never `self.rng`), so folding it in shifts - /// only the pick-table weights, not the scheduler's own pick stream. + /// Plan durable teleports through the fleet handler. DrainHost(usize), + /// Request retirement through the fleet handler. Driven by targeted workloads. + RetireHost(usize), /// ADR 0116 A3: ONE host registers (the store half of the register /// endpoint — a fresh lease written, epoch bumped, any declared /// handoff ended) without the fleet-wide heartbeat sweep of @@ -196,7 +184,7 @@ pub struct Sim { dead_host_cfg: engram_coordinator::dead_host::DeadHostConfig, queue_cfg: engram_coordinator::queue_scanner::QueueScannerConfig, idle_cfg: engram_coordinator::idle_detector::IdleDetectorConfig, - evac_cfg: engram_coordinator::evac_resumer::EvacResumerConfig, + teleport_cfg: engram_coordinator::teleport::TeleportConfig, enable_cfg: engram_coordinator::enable_scanner::EnableScannerConfig, checkpoint_retention_cfg: engram_coordinator::checkpoint_retention::CheckpointRetentionConfig, base_snapshot_retention_cfg: @@ -230,7 +218,7 @@ impl Sim { dead_host_cfg: engram_coordinator::dead_host::DeadHostConfig::default(), queue_cfg: engram_coordinator::queue_scanner::QueueScannerConfig::default(), idle_cfg: engram_coordinator::idle_detector::IdleDetectorConfig::default(), - evac_cfg: engram_coordinator::evac_resumer::EvacResumerConfig::default(), + teleport_cfg: engram_coordinator::teleport::TeleportConfig::default(), enable_cfg: engram_coordinator::enable_scanner::EnableScannerConfig::default(), checkpoint_retention_cfg: engram_coordinator::checkpoint_retention::CheckpointRetentionConfig { @@ -566,8 +554,8 @@ impl Sim { DriverKind::IdleEvictor => { let _ = engram_coordinator::idle_evictor::scanner_run_once(&state).await; } - DriverKind::EvacResumer => { - let _ = engram_coordinator::evac_resumer::run_once(&self.evac_cfg, &state) + DriverKind::Teleport => { + let _ = engram_coordinator::teleport::run_once(&self.teleport_cfg, &state) .await; } DriverKind::EnableScanner => { @@ -1008,7 +996,7 @@ impl Sim { // Deliver op behind, and two delivery-path holes then // livelock a chaos world: // 1. `op_enqueue_and_claim_exclusive` treats ANY queued - // op as a busy lane, so the EvacResumer's claim-or- + // op as a busy lane, so the Teleport's claim-or- // give-up can never acquire an Evacuating session // that holds an undelivered prompt — the evacuation // starves forever while the deliver retries "its @@ -1184,51 +1172,15 @@ impl Sim { self.model.record_destroy_acked(sid); } } - Step::DrainHost(i) => { + Step::DrainHost(i) | Step::RetireHost(i) => { let Some(state) = self.world.replicas.iter().find_map(|r| r.state.clone()) else { return; }; - let host_id = self.world.host_ids[i]; - // The operator drain: the DURABLE cordon (ADR 0047) + evacuate - // each bound session to Evacuating, which the EvacResumer - // driver then re-homes (the #775 dormant leg). The real - // `admin_drain_host` fans the per-session moves out over a - // detached `JoinSet`; that concurrency is UNSIMULABLE — the - // in-flight `blob.put`s complete in nondeterministic order and - // diverge the entropy stream — so here we drive the SAME cordon - // + evict-to-Evacuating pipeline (the drain's ADR 0079 - // fallback) SEQUENTIALLY in deterministic BTreeMap order. The - // full gRPC handler (JoinSet, live-teleport preview, the - // don't-strand guard) is exercised in tests/api_surface.rs. - let _ = state - .services - .meta - .set_host_cordon( - host_id, - Some(engram_core::types::host::CordonOwner::Admin), - None, - ) - .await; - let bound = state - .services - .meta - .list_resident_sandbox_assignments_on_host(host_id) - .await - .unwrap_or_default(); - for (sid, _, st) in bound { - if st != engram_core::types::SessionState::Active { - continue; - } - Self::enqueue_and_drive( - &state, - sid, - OpKind::Evict, - serde_json::json!({ - "target": "evacuating", "allow_park": false, "nominated": false - }), - Some(&format!("drain-evict:{sid}")), - ) - .await; + let host = self.world.host_ids[i]; + if matches!(step, Step::RetireHost(_)) { + crate::workload::api_retire_host(&state, host).await; + } else { + crate::workload::api_drain_host(&state, host).await; } crate::workload::drain_detached().await; } diff --git a/crates/engram-dst/src/workload.rs b/crates/engram-dst/src/workload.rs index 538dcf8be..f63dffa99 100644 --- a/crates/engram-dst/src/workload.rs +++ b/crates/engram-dst/src/workload.rs @@ -142,7 +142,7 @@ pub async fn api_delete(state: &SharedState, session_id: SessionId) -> bool { } /// gRPC `admin_drain_host` (handler-direct, FleetService) — the operator -/// drain that moves bound sessions into Evacuating. +/// drain that schedules durable teleports. pub async fn api_drain_host(state: &SharedState, host_id: HostId) -> bool { let req = grpc_req(app::AdminDrainHostRequest { host_id: host_id.to_string(), @@ -150,6 +150,16 @@ pub async fn api_drain_host(state: &SharedState, host_id: HostId) -> bool { fleet_service(state).admin_drain_host(req).await.is_ok() } +/// Request retirement and schedule teleports through the real handler. +pub async fn api_retire_host(state: &SharedState, host_id: HostId) -> bool { + let req = grpc_req(app::RetireHostRequest { + host_id: host_id.to_string(), + owner: "admin".into(), + reason: "DST retirement".into(), + }); + fleet_service(state).retire_host(req).await.is_ok() +} + /// The axum internal surface for `state`, built fresh (cheap) so a /// replica's Router always reflects its current `SharedState` (rebuilt on /// restart). diff --git a/crates/engram-dst/src/world.rs b/crates/engram-dst/src/world.rs index 9d7e3c008..4cb6e1655 100644 --- a/crates/engram-dst/src/world.rs +++ b/crates/engram-dst/src/world.rs @@ -921,6 +921,15 @@ impl HostClient for SimHostClient { )) } + async fn snapshot_hold( + &self, + id: SandboxId, + fence: SessionFence, + ) -> Result { + self.pause(id, fence).await?; + self.snapshot(id, fence).await + } + async fn snapshot( &self, id: SandboxId, diff --git a/crates/engram-dst/tests/api_surface.rs b/crates/engram-dst/tests/api_surface.rs index bc7c58ebb..5c68acaed 100644 --- a/crates/engram-dst/tests/api_surface.rs +++ b/crates/engram-dst/tests/api_surface.rs @@ -140,7 +140,7 @@ fn api_http_router_real_wire() { /// cordons the host and moves the bound session off Active (toward /// Evacuating / a live rehome). #[test] -fn api_drain_host_cordons_and_evacuates() { +fn api_drain_host_cordons_and_plans() { rt().block_on(async { tokio::time::pause(); let mut sim = Sim::new(9, Profile::Calm).with_faithful_hosts(); @@ -166,20 +166,15 @@ fn api_drain_host_cordons_and_evacuates() { ); workload::drain_detached().await; - let (cordoned, off_active) = sim.world.meta.with_db(|db| { + let (cordoned, planned) = sim.world.meta.with_db(|db| { let cordoned = db.hosts.get(&host).map(|h| h.cordoned).unwrap_or(false); - let off_active = db - .sessions - .get(&sid) - .map(|r| r.session.status != engram_core::types::session::SessionState::Active) - .unwrap_or(true); - (cordoned, off_active) + let planned = db.session_ops.values().any(|op| { + op.session_id == sid && op.kind == engram_core::types::session_op::OpKind::Teleport + }); + (cordoned, planned) }); assert!(cordoned, "drain must durably cordon the host"); - assert!( - off_active, - "drain must move the bound session off Active (evacuation)", - ); + assert!(planned, "drain must plan a durable teleport",); }); } diff --git a/crates/engram-dst/tests/first_sim.rs b/crates/engram-dst/tests/first_sim.rs index 5dd5712f6..3c0e6fa77 100644 --- a/crates/engram-dst/tests/first_sim.rs +++ b/crates/engram-dst/tests/first_sim.rs @@ -113,7 +113,7 @@ fn driver_coverage_is_declared() { DriverKind::OpReclaim, DriverKind::IdleDetector, DriverKind::IdleEvictor, - DriverKind::EvacResumer, + DriverKind::Teleport, DriverKind::EnableScanner, DriverKind::CheckpointRetention, DriverKind::BaseSnapshotRetention, @@ -140,7 +140,7 @@ fn driver_coverage_is_declared() { DriverKind::OpReclaim => &[], DriverKind::IdleDetector => &["idle_detector"], DriverKind::IdleEvictor => &["idle_evictor"], - DriverKind::EvacResumer => &["evac_resumer"], + DriverKind::Teleport => &["teleport"], DriverKind::EnableScanner => &["enable_scanner"], DriverKind::CheckpointRetention => &["checkpoint_retention"], DriverKind::BaseSnapshotRetention => &["base_snapshot_retention"], diff --git a/crates/engram-dst/tests/model_oracle.rs b/crates/engram-dst/tests/model_oracle.rs index df7316312..4ccd4e8dc 100644 --- a/crates/engram-dst/tests/model_oracle.rs +++ b/crates/engram-dst/tests/model_oracle.rs @@ -75,3 +75,66 @@ fn model_oracle_fires_on_a_dropped_live_session_row() { ); }); } + +#[test] +fn capacity_oracle_counts_an_open_teleport_destination() { + use engram_core::traits::{Clock, Entropy, MetadataStore}; + use engram_core::types::teleport::*; + rt().block_on(async { + tokio::time::pause(); + let mut sim = Sim::new(11, Profile::Calm).with_faithful_hosts(); + sim.execute(Step::HostHeartbeats).await; + sim.execute(Step::CreateSession).await; + let session = sim.world.meta.with_db(|db| { + db.sessions + .values() + .find(|s| s.session.status == engram_core::types::SessionState::Active) + .unwrap() + .session + .clone() + }); + let source = session.host_id.unwrap(); + let dest = *sim.world.host_ids.iter().find(|h| **h != source).unwrap(); + let id = engram_core::TeleportId::from(sim.world.entropy.uuid()); + let result = sim + .world + .meta + .teleport_admit(TeleportAdmitRequest { + id, + session_id: session.id, + reason: TeleportReason::Ui, + epoch: sim + .world + .meta + .with_db(|db| db.sessions[&session.id].current_epoch), + candidates: vec![dest], + pinned_dest: Some(dest), + mem_budget_mib: 1, + cpu_budget_vcpus: 1, + max_open_per_dest: 1, + live_capable: false, + }) + .await + .unwrap(); + assert!(matches!(result, TeleportAdmitOutcome::Admitted(_))); + sim.world.meta.with_db_mut(|db| { + db.teleports.get_mut(&id).unwrap().mem_budget_mib = + i64::try_from(db.hosts[&dest].utilization.allocatable_mib).unwrap() + 1; + }); + let mut oracle = engram_dst::invariants::Oracles::default(); + assert_eq!( + oracle.check_step(&sim.world).unwrap_err().invariant, + "placement-accounting" + ); + sim.world.meta.with_db_mut(|db| { + let row = db.teleports.get_mut(&id).unwrap(); + row.phase = TeleportPhase::Aborted; + row.finished_at = Some(sim.world.clock.now_utc()); + db.sessions.get_mut(&session.id).unwrap().session.status = + engram_core::types::SessionState::Active; + }); + oracle + .check_step(&sim.world) + .expect("released teleport reservation"); + }); +} diff --git a/crates/engram-dst/tests/recovery_oracle.rs b/crates/engram-dst/tests/recovery_oracle.rs index 017982f2b..cefb95001 100644 --- a/crates/engram-dst/tests/recovery_oracle.rs +++ b/crates/engram-dst/tests/recovery_oracle.rs @@ -86,3 +86,85 @@ fn recovery_oracle_fires_on_an_unresumable_idle_session() { ); }); } + +#[test] +fn quiescence_rejects_open_teleports_and_retired_host_work() { + use engram_core::traits::{Clock, Entropy, MetadataStore}; + use engram_core::types::teleport::*; + use engram_dst::invariants; + rt().block_on(async { + tokio::time::pause(); + let mut sim = Sim::new(11, Profile::Calm).with_faithful_hosts(); + sim.execute(Step::HostHeartbeats).await; + sim.execute(Step::CreateSession).await; + let session = sim.world.meta.with_db(|db| { + db.sessions + .values() + .find(|s| s.session.status == engram_core::types::SessionState::Active) + .unwrap() + .session + .clone() + }); + let source = session.host_id.unwrap(); + let dest = *sim.world.host_ids.iter().find(|h| **h != source).unwrap(); + let id = engram_core::TeleportId::from(sim.world.entropy.uuid()); + let now = sim.world.clock.now_utc(); + sim.world.meta.with_db_mut(|db| { + db.teleports.insert( + id, + TeleportRow { + id, + session_id: session.id, + kind: TeleportKind::Snapshot, + reason: TeleportReason::RetireHost, + phase: TeleportPhase::Admitted, + source_host_id: source, + source_sandbox_id: session.sandbox_id.unwrap(), + dest_host_id: dest, + dest_sandbox_id: None, + pinned_dest: false, + mem_budget_mib: 1, + cpu_budget_vcpus: 1, + snapshot_id: None, + export_id: None, + live_payload: None, + attempts: 0, + error: None, + created_at: now, + updated_at: now, + finished_at: None, + }, + ); + }); + assert_eq!( + invariants::check_quiescence(&sim.world) + .unwrap_err() + .invariant, + "quiescence-open-teleport" + ); + sim.world.meta.with_db_mut(|db| { + db.teleports.remove(&id); + db.hosts.get_mut(&source).unwrap().status = engram_core::types::HostStatus::Retired; + }); + assert_eq!( + invariants::check_quiescence(&sim.world) + .unwrap_err() + .invariant, + "quiescence-retired-host-owned" + ); + sim.world.meta.with_db_mut(|db| { + db.sessions.get_mut(&session.id).unwrap().session.host_id = None; + }); + sim.world + .meta + .record_sandbox_tombstone(source, session.sandbox_id.unwrap(), Some(session.id)) + .await + .unwrap(); + assert_eq!( + invariants::check_quiescence(&sim.world) + .unwrap_err() + .invariant, + "quiescence-retired-host-owned" + ); + }); +} diff --git a/crates/engram-dst/tests/regression_seeds.rs b/crates/engram-dst/tests/regression_seeds.rs index 30a1bc206..7e320b8c7 100644 --- a/crates/engram-dst/tests/regression_seeds.rs +++ b/crates/engram-dst/tests/regression_seeds.rs @@ -368,7 +368,7 @@ fn issue_790_evict_resume_snapshot_safety_faithful() { /// Issue #800: the RESERVED evac-placement over-reservation, exposed by /// folding the operator-drain verb into the swarm (ADR 0098 Phase 3 wave -/// 5). Before the fix, `evac_resumer → evacuate_dead_source → +/// 5). Before the fix, `teleport → evacuate_dead_source → /// pick_for_session` was capacity-SOFT: a drain-driven wave of evacuations /// bound measured-FULL survivors, driving Σ reserved > allocatable — the /// #722/#795 over-reservation class on the EVAC leg, which #795's @@ -406,7 +406,7 @@ fn seed_33058255_quiescence_waits_for_missing_resident_reconcile() { /// Nightly seed 33058131: a drain evict acknowledged its source-host /// destroy, atomically detached the coordinator binding, and finished its /// op while the host-side teardown effect was still deferred. The next -/// EvacResumer claim saw an unowned `Evacuating` row and restored a second +/// Teleport claim saw an unowned `Evacuating` row and restored a second /// sandbox on a peer, violating ADR 0090 single ownership at step 79. /// /// The fix keeps the outgoing sandbox bound through `Evacuating`. The diff --git a/crates/engram-dst/tests/workload_verbs.rs b/crates/engram-dst/tests/workload_verbs.rs index 8cb17aa76..9648dc1ce 100644 --- a/crates/engram-dst/tests/workload_verbs.rs +++ b/crates/engram-dst/tests/workload_verbs.rs @@ -98,20 +98,15 @@ fn workload_verbs_drive_real_handlers() { .position(|h| *h == host2) .expect("bound host is a known host"); sim.execute(Step::DrainHost(host2_idx)).await; - let (cordoned, off_active) = sim.world.meta.with_db(|db| { + let (cordoned, planned) = sim.world.meta.with_db(|db| { let cordoned = db.hosts.get(&host2).map(|h| h.cordoned).unwrap_or(false); - let off_active = db - .sessions - .get(&sid2) - .map(|r| r.session.status != SessionState::Active) - .unwrap_or(true); - (cordoned, off_active) + let planned = db.session_ops.values().any(|op| { + op.session_id == sid2 && op.kind == engram_core::types::session_op::OpKind::Teleport + }); + (cordoned, planned) }); assert!(cordoned, "Step::DrainHost must cordon the drained host"); - assert!( - off_active, - "Step::DrainHost must move its bound session off Active (evacuation)", - ); + assert!(planned, "Step::DrainHost must plan a durable teleport",); // And the whole thing still converges once the fleet heals (the // heal uncordons the drained host so the evacuated leg re-homes). @@ -233,3 +228,58 @@ fn model_oracle_fires_when_a_confirmed_title_is_repainted() { ); }); } + +#[test] +fn retire_host_workload_moves_then_grants() { + use engram_dst::DriverKind; + rt().block_on(async { + tokio::time::pause(); + let mut sim = Sim::new(11, Profile::Calm).with_faithful_hosts(); + let sid = boot_one_active(&mut sim).await; + let source = sim + .world + .meta + .with_db(|db| db.sessions[&sid].session.host_id.unwrap()); + let index = sim + .world + .host_ids + .iter() + .position(|h| *h == source) + .unwrap(); + sim.execute(Step::RetireHost(index)).await; + assert!(sim + .world + .meta + .with_db(|db| db.hosts[&source].retire_requested_at.is_some())); + for _ in 0..20 { + sim.execute(Step::HostHeartbeats).await; + for driver in [ + DriverKind::Teleport, + DriverKind::SessionOps, + DriverKind::Reconcile, + ] { + sim.execute(Step::Driver(0, driver)).await; + } + sim.execute(Step::AdvanceTime(std::time::Duration::from_secs(2))) + .await; + if sim + .world + .meta + .with_db(|db| db.hosts[&source].retired_at.is_some()) + { + break; + } + } + sim.world.meta.with_db(|db| { + assert!( + db.hosts[&source].retired_at.is_some(), + "retirement did not converge: {:?}", + db.teleports + ); + assert_eq!(db.sessions[&sid].session.status, SessionState::Active); + assert_ne!(db.sessions[&sid].session.host_id, Some(source)); + assert!(db.teleports.values().any(|t| t.session_id == sid + && t.phase == engram_core::types::teleport::TeleportPhase::Done)); + }); + }); +} diff --git a/crates/engram-host-agent/src/grpc_server.rs b/crates/engram-host-agent/src/grpc_server.rs index 8c47ecd92..c59fcb387 100644 --- a/crates/engram-host-agent/src/grpc_server.rs +++ b/crates/engram-host-agent/src/grpc_server.rs @@ -362,6 +362,30 @@ impl HostService for HostServiceImpl { .await } + async fn snapshot_hold( + &self, + req: Request, + ) -> Result, Status> { + let span = tracing::info_span!("host.snapshot_hold"); + link_remote_parent(&span, &req); + check_wire_version(&req)?; + async move { + let r = req.into_inner(); + let fence = self.check_session_epoch(&r.session_id, r.fencing_epoch)?; + let id = decode_sandbox_id(&r.uuid)?; + let metadata = self + .inner + .snapshot_hold(id, fence) + .await + .map_err(sandbox_to_status)?; + Ok(Response::new(SnapshotResponse { + metadata_bincode: encode_bincode(&metadata, "SnapshotMetadata")?, + })) + } + .instrument(span) + .await + } + async fn snapshot_begin( &self, req: Request, diff --git a/crates/engram-host-agent/src/host_client.rs b/crates/engram-host-agent/src/host_client.rs index 6a2377de0..b730e296e 100644 --- a/crates/engram-host-agent/src/host_client.rs +++ b/crates/engram-host-agent/src/host_client.rs @@ -136,6 +136,14 @@ impl HostClient for LocalHostClient { self.sandbox.snapshot(id).await } + async fn snapshot_hold( + &self, + id: SandboxId, + _fence: SessionFence, + ) -> Result { + self.sandbox.snapshot_hold(id, false).await + } + async fn snapshot_begin( &self, id: SandboxId, diff --git a/crates/engram-host-agent/src/lib.rs b/crates/engram-host-agent/src/lib.rs index 3e921ec5b..2d3fea262 100644 --- a/crates/engram-host-agent/src/lib.rs +++ b/crates/engram-host-agent/src/lib.rs @@ -639,9 +639,14 @@ impl HostAgent { } }); } - migration::MigrationRole::PostCopySource => { - tracing::warn!(%sandbox_id, - "reattached post-copy SOURCE: staying paused under the ownership rule (never self-resumes)"); + // A held snapshot source has the same contract as a frozen + // post-copy source: its state may already run elsewhere, so + // it never self-resumes; the coordinator's rollback or + // release ends it, and the ownership rule reaps a leftover. + migration::MigrationRole::PostCopySource + | migration::MigrationRole::HeldSource => { + tracing::warn!(%sandbox_id, ?role, + "reattached frozen SOURCE: staying paused under the ownership rule (never self-resumes)"); pooled.note_migration_role(sandbox_id, Some(role)); let pooled_for_src = pooled.clone(); let coord_for_src: Arc = diff --git a/crates/engram-host-agent/src/migration.rs b/crates/engram-host-agent/src/migration.rs index 70c4d9caa..368ad01a4 100644 --- a/crates/engram-host-agent/src/migration.rs +++ b/crates/engram-host-agent/src/migration.rs @@ -25,7 +25,9 @@ use std::time::Duration; use dashmap::DashMap; use engram_chunk_store::manifest::ChunkHash; use engram_core::SandboxId; -use engram_sandbox_firecracker::sandbox_manifest::{ROLE_POST_COPY_DEST, ROLE_POST_COPY_SOURCE}; +use engram_sandbox_firecracker::sandbox_manifest::{ + ROLE_HELD_SOURCE, ROLE_POST_COPY_DEST, ROLE_POST_COPY_SOURCE, +}; /// No commit/abort within this window triggers the coordinator /// ownership check (see module docs). The clock runs from the LAST @@ -46,6 +48,11 @@ pub const EXPORT_TTL: Duration = Duration::from_secs(120); pub enum MigrationRole { PostCopySource, PostCopyDest, + /// ADR 0123 B: a snapshot-kind source held paused after its capture + /// (`snapshot_hold`). Like a post-copy source it never self-resumes + /// and is never a checkpoint candidate; the coordinator's rollback + /// (`resume`) or release (`destroy`) ends it. + HeldSource, } impl MigrationRole { @@ -53,6 +60,7 @@ impl MigrationRole { match self { Self::PostCopySource => ROLE_POST_COPY_SOURCE, Self::PostCopyDest => ROLE_POST_COPY_DEST, + Self::HeldSource => ROLE_HELD_SOURCE, } } @@ -60,6 +68,7 @@ impl MigrationRole { match s { ROLE_POST_COPY_SOURCE => Some(Self::PostCopySource), ROLE_POST_COPY_DEST => Some(Self::PostCopyDest), + ROLE_HELD_SOURCE => Some(Self::HeldSource), _ => None, } } @@ -517,7 +526,11 @@ mod tests { #[test] fn migration_role_round_trips_persistence_strings() { - for role in [MigrationRole::PostCopySource, MigrationRole::PostCopyDest] { + for role in [ + MigrationRole::PostCopySource, + MigrationRole::PostCopyDest, + MigrationRole::HeldSource, + ] { assert_eq!(MigrationRole::parse(role.as_str()), Some(role)); } assert_eq!(MigrationRole::parse("garbage"), None); @@ -683,6 +696,7 @@ fn migration_role_wire_names_agree() { for (role, name) in [ (MigrationRole::PostCopySource, ROLE_POST_COPY_SOURCE), (MigrationRole::PostCopyDest, ROLE_POST_COPY_DEST), + (MigrationRole::HeldSource, ROLE_HELD_SOURCE), ] { assert_eq!(role.as_str(), name); assert_eq!(MigrationRole::parse(name), Some(role)); diff --git a/crates/engram-host-agent/src/pooled_backend.rs b/crates/engram-host-agent/src/pooled_backend.rs index e1a87e8e8..f432b453a 100644 --- a/crates/engram-host-agent/src/pooled_backend.rs +++ b/crates/engram-host-agent/src/pooled_backend.rs @@ -395,6 +395,7 @@ struct CaptureUnwind { Arc>, )>, defused: bool, + resume_on_drop: bool, } impl CaptureUnwind { @@ -411,6 +412,7 @@ impl CaptureUnwind { disk_seal: None, presetup_restore: None, defused: true, + resume_on_drop: true, } } @@ -433,6 +435,7 @@ impl Drop for CaptureUnwind { return; } let inner = self.inner.clone(); + let resume_on_drop = self.resume_on_drop; let id = self.id; let disk_backend = self.disk_backend.take(); let fenced = self.fenced; @@ -479,6 +482,7 @@ impl Drop for CaptureUnwind { backend.set_migration_fence(false); } } + if !resume_on_drop { return; } if let Err(e) = inner.resume(id).await { tracing::warn!( sandbox_id = %id, @@ -808,6 +812,13 @@ pub struct PooledBackend { /// restart between snapshot and commit/abort leaves an orphan dir /// — small bounded leak we accept until a host-side janitor lands). inflight_snapshots: Arc>, + /// ADR 0123 B: `snapshot_hold` keeps a captured source paused and + /// records the swap policy it disarmed; `resume` re-arms it once and + /// `destroy` clears it. + held_swap_policy: DashMap, + /// The metadata of the snapshot a held source produced, for the + /// coordinator's `snapshot_wait` after the hold. + held_snapshots: DashMap, /// ADR 0016 Phase A: unix-ms timestamp of the last successful /// `snapshot(sandbox_id)` per sandbox. Read by `cow_state` to /// populate the memory-tier RPO field. `0`/absent = never @@ -1556,6 +1567,14 @@ impl PooledBackend { fn spawn_swap_rearm(self: &Arc, id: SandboxId) { let this = Arc::clone(self); tokio::spawn(async move { + // Under the capture lock: a re-arm that lands inside another + // capture's disarm→pause window would put swap PTEs into the + // image that capture is about to take (ADR 0112 D3). A hold + // or a destroy that arrived first wins and the re-arm is moot. + let _capture = this.capture_lock(id).lock_owned().await; + if this.held_swap_policy.contains_key(&id) { + return; + } let script = "for d in /sys/block/vd*; do n=$(basename $d); \ [ \"$n\" != vda ] && [ \"$(cat $d/ro)\" = 0 ] && \ swapon /dev/$n; done"; @@ -2259,15 +2278,23 @@ impl PooledBackend { if post_copy_dest { migration_roles.insert(new_id, crate::migration::MigrationRole::PostCopyDest); } - state + if let Err(error) = state .backend .relocate_dirty_file(&dirty_root.join(format!("{new_id}.cache"))) .await - .map_err(|error| { - SandboxError::Vm( - format!("move restored sandbox dirty file into place: {error}").into(), - ) - })?; + { + // The VM is live on a device this task is about to + // drop. Tear the VM down FIRST so the device never + // disconnects under a running Firecracker (the + // cross-session I/O hazard this task exists to avoid). + if let Err(e) = inner.destroy(new_id).await { + tracing::warn!(sandbox_id = %new_id, error = %e, + "destroy after a failed dirty-file relocation failed"); + } + return Err(SandboxError::Vm( + format!("move restored sandbox dirty file into place: {error}").into(), + )); + } // ADR 0016 Phase B commit 5: post-restore wiring. The new // sandbox_id is only known here; install it into // `nbd_sandboxes` together with the FlushScheduler so the @@ -2378,6 +2405,8 @@ impl PooledBackend { // (`with_nbd_pool`) — a pool-less host has nothing to classify, // so its residue report is vacuously known-empty. nbd_residue_known: Arc::new(std::sync::atomic::AtomicBool::new(true)), + held_swap_policy: DashMap::new(), + held_snapshots: DashMap::new(), inflight_snapshots: Arc::new(DashMap::new()), last_snapshot_unix_ms: Arc::new(DashMap::new()), checkpoint_pacing: Arc::new(DashMap::new()), @@ -2652,7 +2681,9 @@ impl PooledBackend { &self, id: SandboxId, ) -> Result { - let (capture_guard, cap) = self.capture_phase(id, SwapDisarmPolicy::Terminal).await?; + let (capture_guard, cap) = self + .capture_phase(id, SwapDisarmPolicy::Terminal, false) + .await?; self.spawn_trace_publish(id); // ADR 0116 C3: the cold-base seed's multi-GiB upload is the // canonical background bulk producer — nested-capped so it can @@ -2816,6 +2847,7 @@ impl PooledBackend { &self, id: SandboxId, swap_policy: SwapDisarmPolicy, + hold: bool, ) -> Result<(tokio::sync::OwnedMutexGuard<()>, SnapshotCapture), SandboxError> { let capture_lock = self.capture_lock(id); // ADR 0038 B0: time the lock wait — the gridlock signal. With @@ -2823,6 +2855,11 @@ impl PooledBackend { // here is an eviction/drain blocked on an in-flight capture. let lock_wait = crate::time_source::metrics_now(); let capture_guard = capture_lock.lock_owned().await; + if self.held_swap_policy.contains_key(&id) { + return Err(SandboxError::Snapshot( + "snapshot hold already in progress; resume or destroy required".into(), + )); + } metrics::histogram!(crate::metrics::SNAPSHOT_CAPTURE_LOCK_WAIT_SECONDS) .record(lock_wait.elapsed().as_secs_f64()); // 2026-08-03 `chain_poisoned` alert: the SIGTERM capture-quiesce @@ -2988,6 +3025,17 @@ impl PooledBackend { // the (memory, disk, event-log) triple — the coord resolves // the session_events cursor as "last event at or before this" // when it records the checkpoint. + if hold { + self.held_swap_policy.insert(id, swap_disarm); + // Resume now owns re-arm, even if this capture is cancelled. + swap_rearm.target.take(); + // Persisted BEFORE the pause: a host-agent that restarts while + // the move is open re-adopts this VM as a frozen source (never + // self-resumes, never a checkpoint candidate) instead of a + // running guest the periodic capture would un-pause. + self.set_migration_role(id, Some(crate::migration::MigrationRole::HeldSource)) + .await?; + } let paused_at = self.clock.now_utc(); self.inner .pause(id) @@ -3005,6 +3053,7 @@ impl PooledBackend { // `flush_upload` takes it over (covering the cancellation gap in // `snapshot_begin` between this return and the finisher spawn). let mut unwind = CaptureUnwind::new(self.inner.clone(), id); + unwind.resume_on_drop = !hold; unwind.arm(); // ADR 0038 B3: under the pause, only DRAIN the dirty buffer @@ -3104,7 +3153,9 @@ impl PooledBackend { // resume path (chain seeded → diff). let snap_type = if chain_prev.is_some() { "diff" } else { "full" }; let create_start = crate::time_source::metrics_now(); - let create_res = if chain_prev.is_some() { + let create_res = if hold { + self.inner.snapshot_hold(id, chain_prev.is_some()).await + } else if chain_prev.is_some() { self.inner.snapshot_diff(id).await } else { self.inner.snapshot(id).await @@ -4826,7 +4877,7 @@ impl PooledBackend { swap_policy: SwapDisarmPolicy, class: engram_chunk_store::UploadClass, ) -> Result { - let (_capture_guard, cap) = self.capture_phase(id, swap_policy).await?; + let (_capture_guard, cap) = self.capture_phase(id, swap_policy, false).await?; // Lift the per-jail working-set trace into the blob store under the // session-canonical key so the next resume prefaults it (#517 keyed // the replay; the handler can't publish it itself — SIGKILLed on @@ -6689,8 +6740,8 @@ impl SnapshotFinisher { let dest = cap.dest; let chain_prev = cap.chain_prev; let paused_at = cap.paused_at; - // Issue #202: the finisher now owns the capture. The guest is - // running (inner.snapshot resumed it) and `flush_upload` below + // Issue #202: the finisher now owns the capture. Ordinary snapshots + // resume; snapshot_hold stays paused. `flush_upload` below // owns the drained chunks' re-queue on its own error path, so // take the pending out of the guard and defuse it — from here // the guard's resume/requeue must NOT fire. @@ -6708,9 +6759,9 @@ impl SnapshotFinisher { // coordinator's own lifecycle events (median 4, prod evidence). metadata.paused_at = Some(paused_at); let post = async { - // ADR 0038 B3: the guest has resumed (inner.snapshot above - // brought it back). Upload the drained disk chunks to GCS + - // publish the manifest now — OFF the frozen-guest path. The + // ADR 0038 B3: upload the drained disk chunks and publish the + // manifest. Ordinary snapshots have resumed; snapshot_hold keeps + // the source paused throughout this upload. The // operation scope makes the chunk uploads attach `chunk.flush` // spans to the snapshot op's trace. Awaited here (before the // snapshot is recorded) so the recorded `disk_manifest` @@ -7845,6 +7896,25 @@ impl SandboxBackend for PooledBackend { self.inner.open_guest_stream(id, port).await } + async fn snapshot_hold( + &self, + id: SandboxId, + _diff: bool, + ) -> Result { + if let Some(metadata) = self.held_snapshots.get(&id) { + return Ok(metadata.clone()); + } + let (_guard, cap) = self + .capture_phase(id, SwapDisarmPolicy::Terminal, true) + .await?; + let metadata = self + .finisher_with_class(engram_chunk_store::UploadClass::Foreground) + .finish(id, cap) + .await?; + self.held_snapshots.insert(id, metadata.clone()); + Ok(metadata) + } + async fn snapshot(&self, id: SandboxId) -> Result { // ADR 0112: trait callers are the drain / operator / base-bake // flavors — Terminal disarm. The periodic checkpoint reaches @@ -7904,7 +7974,9 @@ impl SandboxBackend for PooledBackend { )); }; - let (capture_guard, cap) = self.capture_phase(id, SwapDisarmPolicy::Terminal).await?; + let (capture_guard, cap) = self + .capture_phase(id, SwapDisarmPolicy::Terminal, false) + .await?; // Publish the working-set trace for the next resume's prefault (see // `spawn_trace_publish`); detached, never blocks the eviction. self.spawn_trace_publish(id); @@ -9158,34 +9230,49 @@ impl SandboxBackend for PooledBackend { if let Some(peer) = self.migrate_peer_server() { peer.remove(export_id); } - // INVARIANT (see `nbd_sandboxes`): clone the Arc and drop the - // guard before the `requeue_*` awaits. - #[cfg(target_os = "linux")] - let backend = self.nbd_sandboxes.get(&id).map(|e| e.backend.clone()); + // The export is consumed: from here the cleanup must run to the + // end even if this request is cancelled, or a retry would see + // NotFound with the fence still raised. A detached task owns it; + // the request only awaits the result. + let inner = self.inner.clone(); + let migration_roles = self.migration_roles.clone(); #[cfg(target_os = "linux")] - if let Some(backend) = backend { - if let Some(pending) = export.disk_pending { - backend.requeue_pending(pending).await; - } - // Disk post-copy: the sealed bytes go back into `dirty` - // so the resumed guest's next flush captures them. - if let Some(sealed) = export.disk_seal.as_deref() { - backend.requeue_postcopy_seal(sealed).await; + let nbd_sandboxes = self.nbd_sandboxes.clone(); + let export_id = export_id.to_string(); + tokio::spawn(async move { + // INVARIANT (see `nbd_sandboxes`): clone the Arc and drop the + // guard before the `requeue_*` awaits. + #[cfg(target_os = "linux")] + let backend = nbd_sandboxes.get(&id).map(|e| e.backend.clone()); + #[cfg(target_os = "linux")] + if let Some(backend) = backend { + if let Some(pending) = export.disk_pending { + backend.requeue_pending(pending).await; + } + // Disk post-copy: the sealed bytes go back into `dirty` + // so the resumed guest's next flush captures them. + if let Some(sealed) = export.disk_seal.as_deref() { + backend.requeue_postcopy_seal(sealed).await; + } + backend.set_migration_fence(false); } - backend.set_migration_fence(false); - } - let snapshot_dir = export.snapshot_dir.clone(); - let _ = fs::remove_dir_all(&snapshot_dir).await; - drop(export.capture_guard); - self.inner - .resume(id) - .await - .map_err(|e| SandboxError::Snapshot(format!("migration abort resume: {e}")))?; - // Last: the drained bytes are back, the fence is down, and the - // guest runs again whether or not this manifest write succeeds. - self.set_migration_role(id, None).await?; - tracing::info!(sandbox_id = %id, "migration aborted; guest resumed in place (ADR 0045 C1)"); - Ok(()) + let snapshot_dir = export.snapshot_dir.clone(); + let _ = fs::remove_dir_all(&snapshot_dir).await; + drop(export.capture_guard); + inner + .resume(id) + .await + .map_err(|e| SandboxError::Snapshot(format!("migration abort resume: {e}")))?; + // Last: the drained bytes are back, the fence is down, and the + // guest runs again whether or not this manifest write succeeds + // (`resume` retries the clear). In-memory first, then the manifest. + migration_roles.remove(&id); + inner.set_manifest_migration_role(id, None).await?; + tracing::info!(sandbox_id = %id, export_id, "migration aborted; guest resumed in place (ADR 0045 C1)"); + Ok(()) + }) + .await + .map_err(|e| SandboxError::Snapshot(format!("migration abort task: {e}")))? } async fn commit_snapshot(&self, id: SandboxId) -> Result<(), SandboxError> { @@ -9333,7 +9420,19 @@ impl SandboxBackend for PooledBackend { )); } } - self.inner.resume(id).await + self.inner.resume(id).await?; + if let Some((_, policy)) = self.held_swap_policy.remove(&id) { + self.held_snapshots.remove(&id); + if policy == SwapDisarm::Disarmed { + if let Some(backend) = self.self_ref.get().and_then(std::sync::Weak::upgrade) { + backend.spawn_swap_rearm(id); + } + } + } + // A resumed guest carries no source role. Idempotent, and the retry + // path for a role clear that failed inside an earlier abort or hold. + self.set_migration_role(id, None).await?; + Ok(()) } async fn restore(&self, metadata: SnapshotMetadata) -> Result { @@ -9497,6 +9596,9 @@ impl SandboxBackend for PooledBackend { } async fn destroy(&self, id: SandboxId) -> Result<(), SandboxError> { + // A held source forgets its swap policy and metadata with the VM. + self.held_swap_policy.remove(&id); + self.held_snapshots.remove(&id); self.teardowns .run_or_join(id, || { let session_bindings = self.session_bindings.clone(); @@ -11384,7 +11486,10 @@ mod tests { .unwrap(); pooled.quiesce_captures_for_shutdown(); - let err = match pooled.capture_phase(id, SwapDisarmPolicy::Terminal).await { + let err = match pooled + .capture_phase(id, SwapDisarmPolicy::Terminal, false) + .await + { Ok(_) => panic!("a quiesced capture must refuse"), Err(e) => e, }; @@ -15984,7 +16089,8 @@ mod tests { let tmp = tempfile::tempdir().unwrap(); let (inner, release) = custody_backend(tmp.path(), false); let pooled = PooledBackend::new(inner.clone()); - let mut capture = Box::pin(pooled.capture_phase(inner.id, SwapDisarmPolicy::Periodic)); + let mut capture = + Box::pin(pooled.capture_phase(inner.id, SwapDisarmPolicy::Periodic, false)); assert!(futures::poll!(&mut capture).is_pending()); let agent = AgentSpec { argv: vec![], @@ -17267,4 +17373,150 @@ mod tests { ); } } + struct HoldBackend { + inner: FakeCaptureBackend, + paused: std::sync::atomic::AtomicBool, + rearms: std::sync::atomic::AtomicUsize, + } + #[async_trait] + impl SandboxBackend for HoldBackend { + async fn create(&self, spec: SandboxSpec) -> Result { + self.inner.create(spec).await + } + async fn exec_stream( + &self, + id: SandboxId, + cmd: ExecRequest, + ) -> Result { + use std::sync::atomic::Ordering; + assert!( + !self.paused.load(Ordering::SeqCst), + "guest exec while paused" + ); + let script = cmd.command.join(" "); + if script.contains("swapon") { + self.rearms.fetch_add(1, Ordering::SeqCst); + } + Ok(ExecStream { + sandbox_id: id, + exec_id: "swap".into(), + events: Box::pin(futures::stream::iter([ + engram_core::types::sandbox::ExecEvent::Stdout(bytes::Bytes::from_static( + b"SwapTotal: 1024 kB\nSwapFree: 1024 kB\nMemAvailable: 65536 kB\n", + )), + engram_core::types::sandbox::ExecEvent::Exit(Some(0)), + ])), + }) + } + fn swap_mib(&self, _: SandboxId) -> Option { + Some(1) + } + async fn pause(&self, _: SandboxId) -> Result<(), SandboxError> { + self.paused.store(true, std::sync::atomic::Ordering::SeqCst); + Ok(()) + } + async fn resume(&self, _: SandboxId) -> Result<(), SandboxError> { + self.paused + .store(false, std::sync::atomic::Ordering::SeqCst); + Ok(()) + } + async fn snapshot(&self, _: SandboxId) -> Result { + panic!("hold must not call the capture that resumes") + } + async fn snapshot_hold( + &self, + id: SandboxId, + _: bool, + ) -> Result { + assert!(self.paused.load(std::sync::atomic::Ordering::SeqCst)); + self.inner.snapshot(id).await + } + fn snapshot_path_for(&self, id: engram_core::SnapshotId) -> PathBuf { + self.inner.snapshot_path_for(id) + } + async fn restore(&self, meta: SnapshotMetadata) -> Result { + self.inner.restore(meta).await + } + async fn destroy(&self, id: SandboxId) -> Result<(), SandboxError> { + self.inner.destroy(id).await + } + async fn list(&self) -> Result, SandboxError> { + self.inner.list().await + } + } + fn held_fixture() -> (tempfile::TempDir, Arc, Arc) { + let dir = tempfile::tempdir().unwrap(); + let inner = Arc::new(HoldBackend { + inner: FakeCaptureBackend { + payload: vec![0; 4096], + staging_root: dir.path().join("snapshots"), + destroy_calls: Arc::new(PlMutex::new(Vec::new())), + }, + paused: std::sync::atomic::AtomicBool::new(false), + rearms: std::sync::atomic::AtomicUsize::new(0), + }); + let pooled = Arc::new(PooledBackend::new(inner.clone())); + pooled.set_self_ref(&pooled); + (dir, pooled, inner) + } + #[tokio::test] + async fn snapshot_hold_stays_paused_without_swap_rearm() { + let (_dir, pooled, inner) = held_fixture(); + let id = SandboxId::new(); + let first = pooled.snapshot_hold(id, false).await.unwrap(); + let retry = pooled.snapshot_hold(id, false).await.unwrap(); + assert_eq!( + first.id, retry.id, + "a lost response reuses the held capture" + ); + tokio::task::yield_now().await; + assert!(inner.paused.load(std::sync::atomic::Ordering::SeqCst)); + assert_eq!(inner.rearms.load(std::sync::atomic::Ordering::SeqCst), 0); + assert_eq!( + *pooled.held_swap_policy.get(&id).unwrap(), + SwapDisarm::Disarmed + ); + // The hold is a persisted source role: a host-agent restart + // re-adopts the paused VM as frozen, never as a running guest. + assert_eq!( + pooled.migration_role(id), + Some(crate::migration::MigrationRole::HeldSource) + ); + assert!( + pooled.snapshot(id).await.is_err(), + "periodic capture must not touch a held VM" + ); + } + #[tokio::test] + async fn resume_after_snapshot_hold_rearms_once() { + let (_dir, pooled, inner) = held_fixture(); + let id = SandboxId::new(); + pooled.snapshot_hold(id, false).await.unwrap(); + pooled.resume(id).await.unwrap(); + pooled.resume(id).await.unwrap(); + for _ in 0..20 { + tokio::task::yield_now().await; + } + assert!(!inner.paused.load(std::sync::atomic::Ordering::SeqCst)); + assert_eq!(inner.rearms.load(std::sync::atomic::Ordering::SeqCst), 1); + assert!(!pooled.held_swap_policy.contains_key(&id)); + assert!(!pooled.held_snapshots.contains_key(&id)); + assert_eq!( + pooled.migration_role(id), + None, + "resume clears the held role" + ); + } + #[tokio::test] + async fn destroy_after_snapshot_hold_clears_policy() { + let (_dir, pooled, inner) = held_fixture(); + let id = SandboxId::new(); + pooled.snapshot_hold(id, false).await.unwrap(); + pooled.destroy(id).await.unwrap(); + tokio::task::yield_now().await; + assert!(!pooled.held_swap_policy.contains_key(&id)); + assert!(!pooled.held_snapshots.contains_key(&id)); + assert_eq!(inner.rearms.load(std::sync::atomic::Ordering::SeqCst), 0); + assert_eq!(inner.inner.destroy_calls.lock().as_slice(), &[id]); + } } diff --git a/crates/engram-postgres/src/lib.rs b/crates/engram-postgres/src/lib.rs index 2ddd0b1a8..5179e8aca 100644 --- a/crates/engram-postgres/src/lib.rs +++ b/crates/engram-postgres/src/lib.rs @@ -10,6 +10,8 @@ use engram_core::types::host::{ CordonOwner, HeartbeatAck, RetirementBlocker, RetirementGrant, RetirementStatus, }; +use engram_core::types::ids::TeleportId; +use engram_core::types::teleport::*; use std::sync::Arc; use async_trait::async_trait; @@ -713,11 +715,16 @@ async fn pick_host_2d( UNION ALL SELECT dest_host_id, mem_budget_mib, cpu_budget_vcpus::BIGINT FROM session_teleports WHERE phase IN ({teleport_reserving}) + UNION ALL + SELECT source_host_id, mem_budget_mib, cpu_budget_vcpus::BIGINT + FROM session_teleports WHERE phase IN ({source_reserving}) ) reserved GROUP BY host_id "#, reserving = reserving_states_sql(), teleport_reserving = engram_core::types::teleport::TeleportPhase::reserving_phases_sql(), + source_reserving = + engram_core::types::teleport::TeleportPhase::source_reserving_phases_sql(), ); let res_rows = sqlx::query(&res_sql) .bind(cand) @@ -1108,6 +1115,264 @@ impl PostgresStore { #[async_trait] impl MetadataStore for PostgresStore { + async fn session_binding_generations(&self, id: SessionId) -> Result<(u64, u64), MetaError> { + let r: (i64, i64) = sqlx::query_as( + "SELECT binding_epoch, attached_binding_epoch FROM sessions WHERE id=$1", + ) + .bind(id.as_uuid()) + .fetch_optional(&self.pool) + .await + .map_err(db_err)? + .ok_or(MetaError::NotFound)?; + Ok((r.0 as u64, r.1 as u64)) + } + + async fn teleport_admit( + &self, + req: TeleportAdmitRequest, + ) -> Result { + let mut tx = self.pool.begin().await.map_err(db_err)?; + let Some(session) = sqlx::query("SELECT status, host_id, sandbox_id FROM sessions WHERE id=$1 AND current_epoch=$2 FOR UPDATE") + .bind(req.session_id.as_uuid()).bind(req.epoch).fetch_optional(&mut *tx).await.map_err(db_err)? else { return Ok(TeleportAdmitOutcome::Fenced) }; + let open: bool = sqlx::query_scalar("SELECT EXISTS (SELECT 1 FROM session_teleports WHERE session_id=$1 AND phase NOT IN ('done','aborted','failed'))") + .bind(req.session_id.as_uuid()).fetch_one(&mut *tx).await.map_err(db_err)?; + if open { + return Err(MetaError::Conflict( + "session already has an open teleport".into(), + )); + } + let status = row::parse_session_state_for_lib( + &session.try_get::("status").map_err(db_err)?, + )?; + if status != SessionState::Active { + return Ok(TeleportAdmitOutcome::SessionNotActive(status)); + } + let source: Option = session.try_get("host_id").map_err(db_err)?; + let sandbox: Option = session.try_get("sandbox_id").map_err(db_err)?; + let (Some(source), Some(sandbox)) = (source, sandbox) else { + return Err(MetaError::Conflict( + "active session has no source binding".into(), + )); + }; + let mut candidates: Vec = req + .pinned_dest + .map(|h| vec![h]) + .unwrap_or(req.candidates) + .into_iter() + .map(|h| h.as_uuid()) + .filter(|h| *h != source) + .collect(); + // Lock before counting open moves. Every placement writer uses this order. + sqlx::query("SELECT id FROM hosts WHERE id=ANY($1) ORDER BY id FOR UPDATE") + .bind(&candidates) + .fetch_all(&mut *tx) + .await + .map_err(db_err)?; + let capped: Vec = sqlx::query_scalar("SELECT dest_host_id FROM session_teleports WHERE dest_host_id=ANY($1) AND phase NOT IN ('done','aborted','failed') GROUP BY dest_host_id HAVING COUNT(*) >= $2") + .bind(&candidates).bind(i64::from(req.max_open_per_dest)).fetch_all(&mut *tx).await.map_err(db_err)?; + candidates.retain(|h| req.max_open_per_dest > 0 && !capped.contains(h)); + let Some(dest) = pick_host_2d( + &mut tx, + &candidates, + 0, + req.mem_budget_mib, + req.cpu_budget_vcpus, + ) + .await? + else { + return Ok(TeleportAdmitOutcome::NoFit); + }; + let cpu = i32::try_from(req.cpu_budget_vcpus) + .map_err(|e| MetaError::Serialization(e.to_string()))?; + let now = self.clock.now_utc(); + let kind = if req.live_capable { + TeleportKind::Live + } else { + TeleportKind::Snapshot + }; + let inserted = sqlx::query("INSERT INTO session_teleports (id,session_id,kind,reason,phase,source_host_id,source_sandbox_id,dest_host_id,pinned_dest,mem_budget_mib,cpu_budget_vcpus,created_at,updated_at) VALUES ($1,$2,$3,$4,'admitted',$5,$6,$7,$8,$9,$10,$11,$11) RETURNING to_jsonb(session_teleports) AS row") + .bind(req.id.as_uuid()).bind(req.session_id.as_uuid()).bind(kind.as_str()).bind(req.reason.as_str()).bind(source).bind(sandbox).bind(dest).bind(req.pinned_dest.is_some()).bind(req.mem_budget_mib).bind(cpu).bind(now) + .fetch_one(&mut *tx).await.map_err(|e| if e.as_database_error().is_some_and(|e| e.is_unique_violation()) { MetaError::Conflict("teleport already exists".into()) } else { db_err(e) })?; + let row: TeleportRow = serde_json::from_value(inserted.try_get("row").map_err(db_err)?) + .map_err(|e| MetaError::Serialization(e.to_string()))?; + sqlx::query("UPDATE sessions SET status='evacuating', last_active_at=$2, updated_at=$2 WHERE id=$1 AND status='active'") + .bind(req.session_id.as_uuid()).bind(now).execute(&mut *tx).await.map_err(db_err)?; + append_event_idempotent_tx( + &mut tx, + req.session_id, + &format!("teleport:{}:admitted", req.id), + "status_changed", + serde_json::json!({"type":"status_changed","from":"active","to":"evacuating","at":now}), + now, + ) + .await?; + tx.commit().await.map_err(db_err)?; + Ok(TeleportAdmitOutcome::Admitted(Box::new(row))) + } + + async fn teleport_advance( + &self, + id: TeleportId, + from: TeleportPhase, + to: TeleportPhase, + patch: TeleportPatch, + epoch: i64, + ) -> Result { + if !from.can_transition_to(to) { + return Err(MetaError::Conflict(format!( + "illegal teleport phase: {from:?} -> {to:?}" + ))); + } + let mut tx = self.pool.begin().await.map_err(db_err)?; + if !lock_teleport_session_at_epoch(&mut tx, id, epoch).await? { + return Ok(false); + } + let n = sqlx::query("UPDATE session_teleports r SET phase=$3, dest_sandbox_id=COALESCE($4,r.dest_sandbox_id), snapshot_id=COALESCE($5,r.snapshot_id), export_id=COALESCE($6,r.export_id), live_payload=COALESCE($12,r.live_payload), kind=COALESCE($7,r.kind), error=$8, attempts=r.attempts+1, updated_at=$10, finished_at=CASE WHEN $11 THEN $10 ELSE NULL END FROM sessions s WHERE r.id=$1 AND r.phase=$2 AND s.id=r.session_id AND s.current_epoch=$9") + .bind(id.as_uuid()).bind(from.as_str()).bind(to.as_str()).bind(patch.dest_sandbox_id.map(|v| v.as_uuid())).bind(patch.snapshot_id.map(|v| v.as_uuid())).bind(patch.export_id).bind(patch.kind.map(|v| v.as_str())).bind(patch.error).bind(epoch).bind(self.clock.now_utc()).bind(to.is_terminal()).bind(patch.live_payload).execute(&mut *tx).await.map_err(db_err)?.rows_affected(); + if n == 1 { + append_teleport_finished_tx(&mut tx, id, self.clock.now_utc()).await?; + } + tx.commit().await.map_err(db_err)?; + Ok(n == 1) + } + + async fn teleport_commit(&self, id: TeleportId, epoch: i64) -> Result, MetaError> { + let mut tx = self.pool.begin().await.map_err(db_err)?; + if !lock_teleport_session_at_epoch(&mut tx, id, epoch).await? { + return Ok(None); + } + let result: Option = sqlx::query_scalar("WITH movement AS MATERIALIZED ( + SELECT * FROM session_teleports WHERE id=$1 AND phase='restored' AND dest_sandbox_id IS NOT NULL FOR UPDATE + ), owner AS MATERIALIZED ( + SELECT s.id FROM sessions s JOIN movement r ON s.id=r.session_id + WHERE s.current_epoch=$2 AND s.status='evacuating' AND s.host_id=r.source_host_id AND s.sandbox_id=r.source_sandbox_id FOR UPDATE OF s + ), rebound AS ( + UPDATE sessions s SET host_id=r.dest_host_id, sandbox_id=r.dest_sandbox_id, binding_epoch=s.binding_epoch+1, missing_strikes=0, updated_at=$3 + FROM movement r, owner o WHERE s.id=o.id RETURNING s.binding_epoch + ), advanced AS ( + UPDATE session_teleports SET phase='committed', updated_at=$3, attempts=attempts+1 WHERE id=$1 AND EXISTS (SELECT 1 FROM rebound) RETURNING id + ) SELECT binding_epoch FROM rebound, advanced") + .bind(id.as_uuid()).bind(epoch).bind(self.clock.now_utc()).fetch_optional(&mut *tx).await.map_err(db_err)?; + tx.commit().await.map_err(db_err)?; + if result.is_some() { + self.notify_placement_changed("teleport_committed").await; + } + Ok(result.map(|n| n as u64)) + } + + async fn teleport_release_source( + &self, + id: TeleportId, + epoch: i64, + how: SourceRelease, + ) -> Result { + let mut tx = self.pool.begin().await.map_err(db_err)?; + if !lock_teleport_session_at_epoch(&mut tx, id, epoch).await? { + return Ok(false); + } + // A deleted host row counts as gone: nothing can answer for it. + let row = sqlx::query("UPDATE session_teleports r SET phase='done', finished_at=$3, updated_at=$3, attempts=r.attempts+1 FROM sessions s WHERE r.id=$1 AND r.phase='attached' AND s.id=r.session_id AND s.current_epoch=$2 AND (NOT $4 OR NOT EXISTS (SELECT 1 FROM hosts h WHERE h.id=r.source_host_id AND h.status NOT IN ('dead','retired'))) RETURNING r.source_host_id,r.source_sandbox_id,r.session_id") + .bind(id.as_uuid()).bind(epoch).bind(self.clock.now_utc()).bind(matches!(how,SourceRelease::SourceHostGone)).fetch_optional(&mut *tx).await.map_err(db_err)?; + let Some(row) = row else { return Ok(false) }; + sqlx::query("INSERT INTO sandbox_tombstones (host_id,sandbox_id,session_id,created_at) VALUES ($1,$2,$3,$4) ON CONFLICT DO NOTHING") + .bind(row.try_get::("source_host_id").map_err(db_err)?).bind(row.try_get::("source_sandbox_id").map_err(db_err)?).bind(row.try_get::("session_id").map_err(db_err)?).bind(self.clock.now_utc()).execute(&mut *tx).await.map_err(db_err)?; + append_teleport_finished_tx(&mut tx, id, self.clock.now_utc()).await?; + tx.commit().await.map_err(db_err)?; + Ok(true) + } + + async fn teleport_settle( + &self, + id: TeleportId, + epoch: i64, + settle: TeleportSettle, + ) -> Result { + let mut tx = self.pool.begin().await.map_err(db_err)?; + if !lock_teleport_session_at_epoch(&mut tx, id, epoch).await? { + return Ok(false); + } + let now = self.clock.now_utc(); + let Some(row) = sqlx::query("UPDATE session_teleports SET phase='failed', error=$2, finished_at=$3, updated_at=$3, attempts=attempts+1 WHERE id=$1 AND phase NOT IN ('done','aborted','failed') RETURNING session_id, source_host_id, source_sandbox_id") + .bind(id.as_uuid()).bind(&settle.error).bind(now).fetch_optional(&mut *tx).await.map_err(db_err)? else { + tx.rollback().await.map_err(db_err)?; + return Ok(false); + }; + let session_id = SessionId::from(row.try_get::("session_id").map_err(db_err)?); + if let Some(target) = settle.session { + let current_raw: String = sqlx::query_scalar("SELECT status FROM sessions WHERE id=$1") + .bind(session_id.as_uuid()) + .fetch_one(&mut *tx) + .await + .map_err(db_err)?; + let current = row::parse_session_state_for_lib(¤t_raw)?; + current + .try_transition_to(target) + .map_err(|e| MetaError::Conflict(e.to_string()))?; + sqlx::query( + "UPDATE sessions SET status=$2, sandbox_id=NULL, updated_at=$3 WHERE id=$1", + ) + .bind(session_id.as_uuid()) + .bind(target.as_str()) + .bind(now) + .execute(&mut *tx) + .await + .map_err(db_err)?; + append_event_idempotent_tx( + &mut tx, + session_id, + &format!("teleport:{id}:settled"), + "status_changed", + serde_json::json!({"type":"status_changed","from":current.as_str(),"to":target.as_str(),"at":now}), + now, + ) + .await?; + } + if settle.entomb_source { + sqlx::query("INSERT INTO sandbox_tombstones (host_id,sandbox_id,session_id,created_at) VALUES ($1,$2,$3,$4) ON CONFLICT DO NOTHING") + .bind(row.try_get::("source_host_id").map_err(db_err)?).bind(row.try_get::("source_sandbox_id").map_err(db_err)?).bind(session_id.as_uuid()).bind(now).execute(&mut *tx).await.map_err(db_err)?; + } + append_teleport_finished_tx(&mut tx, id, now).await?; + tx.commit().await.map_err(db_err)?; + if settle.session.is_some() { + self.notify_placement_changed("teleport_settled").await; + } + Ok(true) + } + async fn teleport_abort(&self, id: TeleportId, epoch: i64) -> Result { + let error: Option = + sqlx::query_scalar("SELECT error FROM session_teleports WHERE id=$1") + .bind(id.as_uuid()) + .fetch_optional(&self.pool) + .await + .map_err(db_err)? + .flatten(); + self.teleport_advance( + id, + TeleportPhase::RollingBack, + TeleportPhase::Aborted, + TeleportPatch { + error, + ..Default::default() + }, + epoch, + ) + .await + } + async fn list_open_teleports(&self) -> Result, MetaError> { + let rows: Vec = sqlx::query_scalar("SELECT to_jsonb(r) FROM session_teleports r WHERE phase NOT IN ('done','aborted','failed') ORDER BY created_at,id").fetch_all(&self.pool).await.map_err(db_err)?; + rows.into_iter() + .map(|r| serde_json::from_value(r).map_err(|e| MetaError::Serialization(e.to_string()))) + .collect() + } + async fn open_teleport_for_session( + &self, + sid: SessionId, + ) -> Result, MetaError> { + let row: Option = sqlx::query_scalar("SELECT to_jsonb(r) FROM session_teleports r WHERE session_id=$1 AND phase NOT IN ('done','aborted','failed')").bind(sid.as_uuid()).fetch_optional(&self.pool).await.map_err(db_err)?; + row.map(|r| serde_json::from_value(r).map_err(|e| MetaError::Serialization(e.to_string()))) + .transpose() + } + async fn ping(&self) -> Result<(), MetaError> { sqlx::query("SELECT 1") .execute(&self.pool) @@ -1387,46 +1652,6 @@ impl MetadataStore for PostgresStore { Ok(disposition) } - async fn set_teleport_target( - &self, - id: SessionId, - target: Option, - ) -> Result<(), MetaError> { - // Issue #214: stamp `teleport_target_set_at` whenever a pin is - // set (NOW()), and clear it when the pin is cleared (Some→non-NULL, - // None→NULL), so the two columns are always consistent. The scanner - // ages out a stale pin off this timestamp. - sqlx::query( - "UPDATE sessions \ - SET teleport_target_host_id = $2, \ - teleport_target_set_at = CASE WHEN $2 IS NULL THEN NULL ELSE $3 END \ - WHERE id = $1", - ) - .bind(id.as_uuid()) - .bind(target.map(|h| h.as_uuid())) - .bind(self.clock.now_utc()) - .execute(&self.pool) - .await - .map_err(db_err)?; - Ok(()) - } - - async fn get_teleport_target( - &self, - id: SessionId, - ) -> Result>)>, MetaError> { - let row: Option<(Option, Option>)> = - sqlx::query_as( - "SELECT teleport_target_host_id, teleport_target_set_at \ - FROM sessions WHERE id = $1", - ) - .bind(id.as_uuid()) - .fetch_optional(&self.pool) - .await - .map_err(db_err)?; - Ok(row.and_then(|(t, set_at)| t.map(|u| (HostId(u), set_at)))) - } - async fn insert_broker_token( &self, token: engram_core::types::registry::SessionBrokerToken, @@ -2330,12 +2555,17 @@ impl MetadataStore for PostgresStore { UNION ALL SELECT dest_host_id, mem_budget_mib, cpu_budget_vcpus::BIGINT FROM session_teleports WHERE phase IN ({teleport_reserving}) + UNION ALL + SELECT source_host_id, mem_budget_mib, cpu_budget_vcpus::BIGINT + FROM session_teleports WHERE phase IN ({source_reserving}) ) reserved GROUP BY host_id "#, reserving = reserving_states_sql(), teleport_reserving = engram_core::types::teleport::TeleportPhase::reserving_phases_sql(), + source_reserving = + engram_core::types::teleport::TeleportPhase::source_reserving_phases_sql(), ); let rows: Vec<(uuid::Uuid, i64, i64)> = sqlx::query_as(&sql) .fetch_all(&self.pool) @@ -2484,12 +2714,17 @@ impl MetadataStore for PostgresStore { UNION ALL SELECT dest_host_id, mem_budget_mib, cpu_budget_vcpus::BIGINT FROM session_teleports WHERE phase IN ({teleport_reserving}) + UNION ALL + SELECT source_host_id, mem_budget_mib, cpu_budget_vcpus::BIGINT + FROM session_teleports WHERE phase IN ({source_reserving}) ) reserved GROUP BY host_id "#, reserving = reserving_states_sql(), teleport_reserving = engram_core::types::teleport::TeleportPhase::reserving_phases_sql(), + source_reserving = + engram_core::types::teleport::TeleportPhase::source_reserving_phases_sql(), ); let res_rows = sqlx::query(&res_sql) .bind(&cand) @@ -2846,13 +3081,7 @@ impl MetadataStore for PostgresStore { target.as_str() ))); } - // ADR 0018 commit 12b: entering Evacuating resets - // `evac_attempts` to 0 so a fresh drain (operator or - // dead-host detector) starts the scanner's retry budget - // clean. ADR 0034 mirrors this for Evicting/`evict_attempts`. - // Folded into the same UPDATE that commits the state flip so - // the counters and the state are always consistent. - // + // Entering Evicting resets its retry count in the same update. // Entering Queued stamps the queue columns (ADR 0098 D4 // conformance finding): a bare `transition_session(_, Queued)` // is FSM-legal but used to leave `queued_at`/`queue_origin` @@ -2868,7 +3097,6 @@ impl MetadataStore for PostgresStore { last_active_at = $3, updated_at = $3, sandbox_id = CASE WHEN $4 THEN NULL ELSE sandbox_id END, - evac_attempts = CASE WHEN $2 = 'evacuating' THEN 0 ELSE evac_attempts END, evict_attempts = CASE WHEN $2 = 'evicting' THEN 0 ELSE evict_attempts END, queued_at = CASE WHEN $2 = 'queued' THEN $3 ELSE queued_at END, queue_origin = CASE WHEN $2 = 'queued' @@ -2897,43 +3125,13 @@ impl MetadataStore for PostgresStore { Ok(current) } - /// ADR 0018 commit 12b: scanner sweep query. Indexed via the - /// partial `idx_sessions_evacuating` from migration 0037 so the - /// cost stays flat as the global session row count grows. - async fn list_evacuating_sessions(&self) -> Result, MetaError> { - let rows = sqlx::query( - r#" - SELECT id, status, host_id, sandbox_id, - image_uri, mode, - created_at, last_active_at, - live_disk_manifest_id, live_disk_manifest_version, - evac_attempts - FROM sessions - WHERE status = 'evacuating' - "#, - ) - .fetch_all(&self.pool) - .await - .map_err(db_err)?; - let mut out = Vec::with_capacity(rows.len()); - for r in &rows { - let session = row::session_from_row(r)?; - let attempts: i32 = r - .try_get("evac_attempts") - .map_err(|e| MetaError::Serialization(format!("evac_attempts: {e}")))?; - out.push((session, attempts.max(0) as u32)); - } - Ok(out) - } - async fn list_host_lost_sessions(&self) -> Result, MetaError> { let rows = sqlx::query( r#" SELECT id, status, host_id, sandbox_id, image_uri, mode, created_at, last_active_at, - live_disk_manifest_id, live_disk_manifest_version, - evac_attempts + live_disk_manifest_id, live_disk_manifest_version FROM sessions WHERE status = 'host_lost' "#, @@ -2948,30 +3146,6 @@ impl MetadataStore for PostgresStore { Ok(out) } - /// ADR 0018 commit 12b: atomic `+= 1 RETURNING`. Scanner calls - /// this before each resume attempt so the returned count is - /// the scanner's "this is my Nth try" view; when it crosses - /// the budget threshold, the scanner falls back to Idle. - async fn bump_evac_attempts(&self, session_id: SessionId) -> Result { - let row = sqlx::query( - r#" - UPDATE sessions - SET evac_attempts = evac_attempts + 1 - WHERE id = $1 - RETURNING evac_attempts - "#, - ) - .bind(session_id.as_uuid()) - .fetch_optional(&self.pool) - .await - .map_err(db_err)? - .ok_or(MetaError::NotFound)?; - let attempts: i32 = row - .try_get("evac_attempts") - .map_err(|e| MetaError::Serialization(format!("bump_evac_attempts: {e}")))?; - Ok(attempts.max(0) as u32) - } - /// ADR 0034: eviction-scanner sweep query. Indexed via the /// partial `idx_sessions_evicting` from migration 0050 so the /// cost stays flat as the global session row count grows. @@ -4596,6 +4770,18 @@ impl MetadataStore for PostgresStore { WHERE c.host_id = $1 AND c.stage NOT IN ('done', 'failed') ) + -- ADR 0123 B: a move owns both endpoints while open. The + -- source is unbound after commit until release; the dest + -- is unbound until commit and unnamed until restore + -- returns, so an open move INTO this host protects every + -- unbound sandbox here the way a capture job does. + AND NOT EXISTS ( + SELECT 1 FROM session_teleports t + WHERE t.phase NOT IN ('done', 'aborted', 'failed') + AND (t.source_sandbox_id = r.sandbox_id + OR t.dest_sandbox_id = r.sandbox_id + OR (t.dest_host_id = $1 AND t.dest_sandbox_id IS NULL)) + ) ), pruned AS ( DELETE FROM sandbox_unbound_sightings @@ -4736,6 +4922,7 @@ impl MetadataStore for PostgresStore { FROM sessions WHERE host_id = $1 AND status NOT IN ('completed','failed','dead') + AND NOT EXISTS (SELECT 1 FROM session_teleports r WHERE r.session_id=sessions.id AND r.source_host_id=$1 AND r.phase NOT IN ('done','aborted','failed')) FOR UPDATE ), cleared AS ( @@ -8345,7 +8532,6 @@ impl MetadataStore for PostgresStore { last_active_at = $4, updated_at = $4, sandbox_id = CASE WHEN $5 THEN NULL ELSE sandbox_id END, - evac_attempts = CASE WHEN $2 = 'evacuating' THEN 0 ELSE evac_attempts END, evict_attempts = CASE WHEN $2 = 'evicting' THEN 0 ELSE evict_attempts END, queued_at = CASE WHEN $2 = 'queued' THEN $4 ELSE queued_at END, queue_origin = CASE WHEN $2 = 'queued' @@ -8445,7 +8631,6 @@ impl MetadataStore for PostgresStore { last_active_at = $4, updated_at = $4, sandbox_id = CASE WHEN $5 THEN NULL ELSE sandbox_id END, - evac_attempts = CASE WHEN $2 = 'evacuating' THEN 0 ELSE evac_attempts END, evict_attempts = CASE WHEN $2 = 'evicting' THEN 0 ELSE evict_attempts END, queued_at = CASE WHEN $2 = 'queued' THEN $4 ELSE queued_at END, queue_origin = CASE WHEN $2 = 'queued' @@ -9562,3 +9747,46 @@ async fn append_event_idempotent_tx( .fetch_one(&mut **tx).await.map_err(db_err)?; Ok(Some(row.try_get("idx").map_err(db_err)?)) } + +/// Lock the teleport's session row and answer whether it is still at +/// `epoch`. Every teleport writer takes this lock FIRST (session, then the +/// row), so a reclaim that bumps `current_epoch` either waits for the write +/// or is seen by it: a stale holder can never land a phase change after a +/// successor took the lane. +async fn lock_teleport_session_at_epoch( + tx: &mut sqlx::Transaction<'_, sqlx::Postgres>, + id: TeleportId, + epoch: i64, +) -> Result { + let current: Option = sqlx::query_scalar( + "SELECT s.current_epoch FROM sessions s JOIN session_teleports r ON r.session_id=s.id WHERE r.id=$1 FOR UPDATE OF s", + ) + .bind(id.as_uuid()) + .fetch_optional(&mut **tx) + .await + .map_err(db_err)?; + Ok(current == Some(epoch)) +} + +async fn append_teleport_finished_tx( + tx: &mut sqlx::Transaction<'_, sqlx::Postgres>, + id: TeleportId, + now: DateTime, +) -> Result<(), MetaError> { + let raw:Option=sqlx::query_scalar("SELECT to_jsonb(r) FROM session_teleports r WHERE id=$1 AND phase IN ('done','aborted','failed')").bind(id.as_uuid()).fetch_optional(&mut **tx).await.map_err(db_err)?; + if let Some(raw) = raw { + let row: TeleportRow = + serde_json::from_value(raw).map_err(|e| MetaError::Serialization(e.to_string()))?; + let payload = serde_json::json!({"type":"teleport_finished","teleport_id":row.id,"outcome":row.phase.as_str(),"kind":row.kind,"dest_host_id":row.dest_host_id,"error":row.error,"at":now}); + append_event_idempotent_tx( + tx, + row.session_id, + &format!("teleport:{}:finished", row.id), + "teleport_finished", + payload, + now, + ) + .await?; + } + Ok(()) +} diff --git a/crates/engram-protocol/proto/engram/app/v1/fleet.proto b/crates/engram-protocol/proto/engram/app/v1/fleet.proto index 3c96a43e5..a12fc44f9 100644 --- a/crates/engram-protocol/proto/engram/app/v1/fleet.proto +++ b/crates/engram-protocol/proto/engram/app/v1/fleet.proto @@ -42,8 +42,9 @@ service FleetService { // ADR 0016 Phase B explicit flush trigger // (POST /api/admin/sessions/:id/flush-now). rpc FlushSession(FlushSessionRequest) returns (FlushSessionResponse); - // ADR 0018 async evacuation (POST /api/admin/sessions/:id/evacuate). - rpc EvacuateSession(EvacuateSessionRequest) returns (EvacuateSessionResponse); + // ADR 0123 B: admit one teleport for a session (synchronous admission; + // the durable machine drives it). Replaces the ADR 0018 evacuate route. + rpc TeleportSession(TeleportSessionRequest) returns (TeleportSessionResponse); // The three GC sweeps (ADR 0016 Phase C / ADR 0035 §5 / ADR 0028 // addendum). Each folds today's paired /dry-run + /sweep routes into // one RPC discriminated by `dry_run`. @@ -191,22 +192,12 @@ message AdminDrainHostRequest { string host_id = 1; } -// Mirrors api/admin.rs DrainHostResponse (renamed: this RPC carries the -// admin cordon+evacuate semantics; the soft status flip above keeps the -// plain DrainHost name). HTTP returns 202 — the evac_resumer scanner is -// the actual deliverable. +// Admission plans return immediately; the teleport machine drives each move. message AdminDrainHostResponse { string host_id = 1; - // Session ids now marked Evacuating; the scanner resumes each on a - // peer host within its sweep interval. - repeated string evacuating = 2; - repeated DrainFailure failures = 3; -} - -message DrainFailure { - string session_id = 1; - // Human-readable diagnostic; not machine-parsed, not stable. - string error = 2; + repeated string planned = 2; + repeated string descended = 3; + uint32 skipped = 4; } message CordonHostRequest { @@ -325,24 +316,14 @@ message FlushSessionResponse { optional uint64 manifest_version = 2; } -message EvacuateSessionRequest { +message TeleportSessionRequest { string session_id = 1; - // Reserved for a future operator override; carried on today's HTTP - // body (api/admin.rs EvacuateSessionRequest.target_host) but IGNORED - // by the handler — the scanner picks any non-source host via the - // standard policy. optional string target_host = 2; } - -// Mirrors api/admin.rs EvacuateSessionResponse. HTTP returns 202; the -// evac_resumer scanner completes the transition. -message EvacuateSessionResponse { - string session_id = 1; - // Always "evacuating" on success — the session is paused, - // snapshotted, and the scanner will resume it on a peer within the - // next sweep interval (≤10s default). Watch the session's event - // stream for the Evacuating → Created → Active chain. - string status = 2; +message TeleportSessionResponse { + string teleport_id = 1; + string kind = 2; + string dest_host_id = 3; } message GetFleetDemandRequest {} diff --git a/crates/engram-protocol/proto/host_service.proto b/crates/engram-protocol/proto/host_service.proto index 45150e985..534f821c8 100644 --- a/crates/engram-protocol/proto/host_service.proto +++ b/crates/engram-protocol/proto/host_service.proto @@ -30,6 +30,8 @@ service HostService { rpc DestroySandbox(FencedSandboxRequest) returns (Empty); rpc ListSandboxes(Empty) returns (ListSandboxesResponse); rpc Snapshot(FencedSandboxRequest) returns (SnapshotResponse); + // Capture and retain the paused source until explicit resume or destroy. + rpc SnapshotHold(FencedSandboxRequest) returns (SnapshotResponse); // ADR 0014 issue #1/#2: signal that a just-produced snapshot's // downstream pipeline (record_snapshot → destroy → mark Idle) // fully succeeded; the host clears its in-flight tracking and the diff --git a/crates/engram-protocol/src/grpc_client.rs b/crates/engram-protocol/src/grpc_client.rs index e1e40f2a5..b6d5201b7 100644 --- a/crates/engram-protocol/src/grpc_client.rs +++ b/crates/engram-protocol/src/grpc_client.rs @@ -254,6 +254,21 @@ impl GrpcHostClient { decode_bincode(&resp.metadata_bincode, "SnapshotMetadata") } + pub async fn snapshot_hold( + &self, + id: SandboxId, + fence: SessionFence, + ) -> Result { + let resp = self + .inner + .clone() + .snapshot_hold(fenced_request(id, fence)) + .await + .map_err(grpc_to_sandbox_err)? + .into_inner(); + decode_bincode(&resp.metadata_bincode, "SnapshotMetadata") + } + /// ADR 0045 D5. An `Unimplemented` status from a pre-D5 host-agent /// maps to `InvalidSpec` (same shape as the trait default), which the /// coordinator treats as "fall back to the composed snapshot()". @@ -1600,6 +1615,14 @@ impl HostClient for GrpcHostClient { Self::snapshot(self, id, fence).await } + async fn snapshot_hold( + &self, + id: SandboxId, + fence: SessionFence, + ) -> Result { + Self::snapshot_hold(self, id, fence).await + } + async fn snapshot_begin( &self, id: SandboxId, diff --git a/crates/engram-protocol/src/wire.rs b/crates/engram-protocol/src/wire.rs index 4852d1f47..c05f35844 100644 --- a/crates/engram-protocol/src/wire.rs +++ b/crates/engram-protocol/src/wire.rs @@ -158,7 +158,9 @@ use serde::{Deserialize, Serialize}; // inject/refresh route. Trailing-variant addition: every existing encoding // is unchanged, but a v27 host cannot decode a policy carrying the new // variant, so the roll is lockstep. -pub const WIRE_VERSION: u32 = 28; +// v29 (ADR 0123): SnapshotHold keeps a source paused through teleport commit. +// Older hosts cannot provide this capture contract. Roll coordinator and hosts together. +pub const WIRE_VERSION: u32 = 29; /// gRPC metadata (header) key carrying the caller's [`WIRE_VERSION`] on /// every coord→host request (issue #229). ASCII, lowercase — tonic diff --git a/crates/engram-protocol/tests/wire_golden.rs b/crates/engram-protocol/tests/wire_golden.rs index 63771ec8a..f768822cb 100644 --- a/crates/engram-protocol/tests/wire_golden.rs +++ b/crates/engram-protocol/tests/wire_golden.rs @@ -678,8 +678,9 @@ fn wire_version_pinned() { // inject rail). Existing goldens keep their bytes (trailing-variant // addition); the oauth-user-policy golden is ADDED and the variant index // is pinned at 2. + // 28 -> 29: SnapshotHold adds an RPC; bincode payloads are unchanged. assert_eq!( - WIRE_VERSION, 28, + WIRE_VERSION, 29, "WIRE_VERSION changed — confirm payload goldens were regenerated too" ); } diff --git a/crates/engram-sandbox-firecracker/src/client.rs b/crates/engram-sandbox-firecracker/src/client.rs index bcb48b1a3..0bd60d9f8 100644 --- a/crates/engram-sandbox-firecracker/src/client.rs +++ b/crates/engram-sandbox-firecracker/src/client.rs @@ -266,6 +266,30 @@ impl FirecrackerClient { }) } + /// Capture a paused VM without a resume guard. The caller owns recovery. + pub async fn create_snapshot_held( + &self, + state_path: PathBuf, + mem_path: PathBuf, + snapshot_type: SnapshotType, + ) -> Result { + self.pause().await?; + self.put( + "/snapshot/create", + &SnapshotCreateBody { + snapshot_path: state_path.to_string_lossy().into_owned(), + mem_file_path: mem_path.to_string_lossy().into_owned(), + snapshot_type, + vmstate_only: false, + }, + ) + .await?; + Ok(SnapshotPaths { + state_path, + mem_path, + }) + } + /// ADR 0045 C2 (fork v3): vmstate-only snapshot create — writes the /// state file at `state_path` and skips the guest-memory leg entirely. /// The post-copy blackout primitive: the destination demand-faults the diff --git a/crates/engram-sandbox-firecracker/src/lib.rs b/crates/engram-sandbox-firecracker/src/lib.rs index bb78b00fc..a57a62631 100644 --- a/crates/engram-sandbox-firecracker/src/lib.rs +++ b/crates/engram-sandbox-firecracker/src/lib.rs @@ -5293,7 +5293,7 @@ impl SandboxBackend for FirecrackerBackend { } async fn snapshot(&self, id: SandboxId) -> Result { - self.snapshot_with_type(id, client::SnapshotType::Full) + self.snapshot_with_type(id, client::SnapshotType::Full, false) .await } @@ -5302,8 +5302,21 @@ impl SandboxBackend for FirecrackerBackend { /// artifact is `memory.diff` (sparse, dirty-pages-only; the KVM /// dirty bitmap resets on capture so successive calls chain). /// Requires `FirecrackerConfig::track_dirty_pages`. + async fn snapshot_hold( + &self, + id: SandboxId, + diff: bool, + ) -> Result { + let kind = if diff { + client::SnapshotType::Diff + } else { + client::SnapshotType::Full + }; + self.snapshot_with_type(id, kind, true).await + } + async fn snapshot_diff(&self, id: SandboxId) -> Result { - self.snapshot_with_type(id, client::SnapshotType::Diff) + self.snapshot_with_type(id, client::SnapshotType::Diff, false) .await } @@ -7147,6 +7160,7 @@ impl FirecrackerBackend { &self, id: SandboxId, snapshot_type: client::SnapshotType, + hold: bool, ) -> Result { // Read sandbox state under the dashmap guard, drop guard before // any await so we don't hold the read lock across an HTTP call. @@ -7214,15 +7228,24 @@ impl FirecrackerBackend { // 32 GiB capture). See `snapshot_create_timeout`. let api = FirecrackerClient::new(&socket) .with_timeout(snapshot_create_timeout(spec.memory.max_mib)); - let create_result = match snapshot_type { - client::SnapshotType::Full => api.create_snapshot(&dest).await, - client::SnapshotType::Diff => { - api.create_snapshot_at( - dest.join("state.bin"), - dest.join("memory.diff"), - client::SnapshotType::Diff, - ) + let create_result = if hold { + let name = match snapshot_type { + client::SnapshotType::Full => "memory.bin", + client::SnapshotType::Diff => "memory.diff", + }; + api.create_snapshot_held(dest.join("state.bin"), dest.join(name), snapshot_type) .await + } else { + match snapshot_type { + client::SnapshotType::Full => api.create_snapshot(&dest).await, + client::SnapshotType::Diff => { + api.create_snapshot_at( + dest.join("state.bin"), + dest.join("memory.diff"), + client::SnapshotType::Diff, + ) + .await + } } }; // The create ran `prepare_save`, which queued a vsock diff --git a/crates/engram-sandbox-firecracker/src/sandbox_manifest.rs b/crates/engram-sandbox-firecracker/src/sandbox_manifest.rs index 47b3f0a39..313d0ab89 100644 --- a/crates/engram-sandbox-firecracker/src/sandbox_manifest.rs +++ b/crates/engram-sandbox-firecracker/src/sandbox_manifest.rs @@ -32,6 +32,8 @@ use serde::{Deserialize, Serialize}; pub const ROLE_POST_COPY_SOURCE: &str = "post-copy-source"; /// Durable role for the destination of a post-copy move. pub const ROLE_POST_COPY_DEST: &str = "post-copy-dest"; +/// ADR 0123 B: a snapshot-kind teleport source held paused after capture. +pub const ROLE_HELD_SOURCE: &str = "held-source"; /// Current schema version. Bumped on incompatible changes to the /// on-disk JSON shape. The startup reattach pass refuses to read diff --git a/crates/engram-sandbox-process/src/lib.rs b/crates/engram-sandbox-process/src/lib.rs index 0899c3425..79b47cde9 100644 --- a/crates/engram-sandbox-process/src/lib.rs +++ b/crates/engram-sandbox-process/src/lib.rs @@ -261,6 +261,30 @@ struct ProcessImage { } impl ProcessBackend { + fn signal_sandbox( + &self, + id: SandboxId, + signal: nix::sys::signal::Signal, + ) -> Result<(), SandboxError> { + let state = self.sandboxes.get(&id).ok_or(SandboxError::NotFound)?; + let mut groups: Vec = state.exec_records.iter().filter_map(|r| r.pgid).collect(); + if let Some(slot) = self.agent_children.get(&id) { + if let Some(child) = slot.lock().unwrap().as_ref() { + if let Some(pid) = child.id() { + groups.push(pid); + } + } + } + for group in groups { + let raw = i32::try_from(group).map_err(|e| SandboxError::Vm(e.to_string().into()))?; + match nix::sys::signal::killpg(nix::unistd::Pid::from_raw(raw), signal) { + Ok(()) | Err(nix::errno::Errno::ESRCH) => {} + Err(e) => return Err(SandboxError::Vm(e.to_string().into())), + } + } + Ok(()) + } + pub fn new(work_dir: impl Into) -> Self { let work_dir = work_dir.into(); // Every emulated guest path derives from this root and may be passed @@ -781,6 +805,23 @@ impl SandboxBackend for ProcessBackend { )) } + async fn pause(&self, id: SandboxId) -> Result<(), SandboxError> { + self.signal_sandbox(id, nix::sys::signal::Signal::SIGSTOP) + } + + async fn resume(&self, id: SandboxId) -> Result<(), SandboxError> { + self.signal_sandbox(id, nix::sys::signal::Signal::SIGCONT) + } + + async fn snapshot_hold( + &self, + id: SandboxId, + _diff: bool, + ) -> Result { + self.pause(id).await?; + self.snapshot(id).await + } + async fn snapshot(&self, id: SandboxId) -> Result { let state = self .sandboxes @@ -1138,6 +1179,7 @@ async fn spawn_agent( .stderr(Stdio::from(log_err)) .kill_on_drop(true); + cmd.process_group(0); let child = cmd .spawn() .map_err(|e| SandboxError::Vm(format!("spawn agent `{argv0}`: {e}").into()))?; diff --git a/crates/engram-sandbox-vz/src/backend.rs b/crates/engram-sandbox-vz/src/backend.rs index b547190ee..cc8ed9ac8 100644 --- a/crates/engram-sandbox-vz/src/backend.rs +++ b/crates/engram-sandbox-vz/src/backend.rs @@ -1381,6 +1381,15 @@ impl SandboxBackend for VzBackend { vm.resume().await.map_err(SandboxError::from) } + async fn snapshot_hold( + &self, + id: SandboxId, + _diff: bool, + ) -> Result { + self.pause(id).await?; + self.snapshot(id).await + } + async fn snapshot(&self, id: SandboxId) -> Result { let (vm, spec, rootfs_path, vsock_uds_path, machine_identifier, mac_address, resolved_aux) = { let live = self.sandboxes.get(&id).ok_or(SandboxError::NotFound)?; diff --git a/crates/engram-sim/src/meta/mod.rs b/crates/engram-sim/src/meta/mod.rs index 215a8f810..74671a9da 100644 --- a/crates/engram-sim/src/meta/mod.rs +++ b/crates/engram-sim/src/meta/mod.rs @@ -151,7 +151,6 @@ pub struct SessRow { pub recovery_epoch: i64, pub shell_pinned_until: Option>, pub durable_head: Option, - pub evac_attempts: i32, pub evict_attempts: i32, pub updated_at: DateTime, } @@ -235,11 +234,7 @@ pub struct SimDb { pub oauth_flows: std::collections::BTreeMap, pub session_oauth_bindings: std::collections::BTreeMap, - /// ADR 0045 live-migration teleport pin (PG: `sessions. - /// teleport_target_host_id` + `_set_at`). Present only while a pin is - /// set. The sim runs no teleport workload today, but the boot/rebind - /// path clears the pin, so get/set must round-trip. - pub teleport_targets: std::collections::BTreeMap>)>, + pub session_ops: std::collections::BTreeMap, pub outbox: std::collections::BTreeMap, pub bundle_gc: std::collections::BTreeMap, diff --git a/crates/engram-sim/src/meta/store_impl.rs b/crates/engram-sim/src/meta/store_impl.rs index 1b10e8256..0600024e6 100644 --- a/crates/engram-sim/src/meta/store_impl.rs +++ b/crates/engram-sim/src/meta/store_impl.rs @@ -8,6 +8,8 @@ use engram_core::types::host::{ CordonOwner, HeartbeatAck, HostLeaseState, RetirementBlocker, RetirementGrant, RetirementStatus, ENABLE_MATERIALIZE_LEASE_SECS, }; +use engram_core::types::ids::TeleportId; +use engram_core::types::teleport::*; use std::time::Duration; use async_trait::async_trait; @@ -78,7 +80,15 @@ fn reserved_budgets(db: &SimDb) -> std::collections::BTreeMap std::collections::BTreeMap, +) -> Option<(SessionId, i64)> { + let r = db.teleports.get(&id)?; + if !r.phase.is_terminal() { + return None; + } + let sid = r.session_id; + let payload = serde_json::json!({"type":"teleport_finished","teleport_id":r.id,"outcome":r.phase.as_str(),"kind":r.kind,"dest_host_id":r.dest_host_id,"error":r.error,"at":now,"idempotency_key":format!("teleport:{}:finished",r.id)}); + let s = db.sessions.get_mut(&sid).expect("teleport session"); + let idx = s.next_event_idx; + s.next_event_idx += 1; + s.updated_at = now; + s.session.last_event_at = Some(now); + let recovery_epoch = s.recovery_epoch; + db.session_events + .entry(sid) + .or_default() + .push(PersistedEvent { + idx, + kind: "teleport_finished".into(), + payload, + created_at: now, + recovery_epoch, + rewound_at: None, + }); + Some((sid, idx)) +} + fn enable_work( db: &SimDb, now: DateTime, @@ -243,6 +284,362 @@ impl SimMetadataStore { #[async_trait] impl MetadataStore for SimMetadataStore { + async fn session_binding_generations(&self, id: SessionId) -> Result<(u64, u64), MetaError> { + self.gate()?; + let db = self.db.lock(); + let r = db.sessions.get(&id).ok_or(MetaError::NotFound)?; + Ok((r.binding_epoch as u64, r.attached_binding_epoch as u64)) + } + + async fn teleport_admit( + &self, + req: TeleportAdmitRequest, + ) -> Result { + self.gate()?; + let now = self.now(); + let mut db = self.db.lock(); + let Some(s) = db + .sessions + .get(&req.session_id) + .filter(|s| s.current_epoch == req.epoch) + else { + return Ok(TeleportAdmitOutcome::Fenced); + }; + if db + .teleports + .values() + .any(|r| r.session_id == req.session_id && !r.phase.is_terminal()) + { + return Err(MetaError::Conflict( + "session already has an open teleport".into(), + )); + } + if s.session.status != SessionState::Active { + return Ok(TeleportAdmitOutcome::SessionNotActive(s.session.status)); + } + let (Some(source), Some(sandbox)) = (s.session.host_id, s.session.sandbox_id) else { + return Err(MetaError::Conflict( + "active session has no source binding".into(), + )); + }; + let candidates: Vec<_> = req + .pinned_dest + .map(|h| vec![h]) + .unwrap_or(req.candidates) + .into_iter() + .filter(|h| { + *h != source + && db + .teleports + .values() + .filter(|r| r.dest_host_id == *h && !r.phase.is_terminal()) + .count() + < req.max_open_per_dest as usize + }) + .collect(); + let Some(dest) = Self::pick_host_2d( + &db, + &candidates, + 0, + req.mem_budget_mib, + req.cpu_budget_vcpus, + ) else { + return Ok(TeleportAdmitOutcome::NoFit); + }; + if db.teleports.contains_key(&req.id) { + return Err(MetaError::Conflict("teleport already exists".into())); + } + let row = TeleportRow { + id: req.id, + session_id: req.session_id, + kind: if req.live_capable { + TeleportKind::Live + } else { + TeleportKind::Snapshot + }, + reason: req.reason, + phase: TeleportPhase::Admitted, + source_host_id: source, + source_sandbox_id: sandbox, + dest_host_id: dest, + dest_sandbox_id: None, + pinned_dest: req.pinned_dest.is_some(), + mem_budget_mib: req.mem_budget_mib, + cpu_budget_vcpus: i32::try_from(req.cpu_budget_vcpus) + .map_err(|e| MetaError::Serialization(e.to_string()))?, + snapshot_id: None, + export_id: None, + live_payload: None, + attempts: 0, + error: None, + created_at: now, + updated_at: now, + finished_at: None, + }; + let s = db + .sessions + .get_mut(&req.session_id) + .expect("locked session"); + s.session.status = SessionState::Evacuating; + s.session.last_active_at = now; + s.session.last_event_at = Some(now); + s.updated_at = now; + let idx = s.next_event_idx; + s.next_event_idx += 1; + let recovery_epoch = s.recovery_epoch; + db.session_events.entry(req.session_id).or_default().push(PersistedEvent {idx, kind:"status_changed".into(),payload:serde_json::json!({"type":"status_changed","from":"active","to":"evacuating","at":now,"idempotency_key":format!("teleport:{}:admitted",req.id)}),created_at:now,recovery_epoch,rewound_at:None}); + db.transition_log.push(super::TransitionLogEntry { + session: req.session_id, + from: SessionState::Active, + to: SessionState::Evacuating, + exempt: false, + }); + db.teleports.insert(req.id, row.clone()); + drop(db); + self.notify( + "session_events", + format!("{{\"session_id\":\"{}\",\"idx\":{idx}}}", req.session_id), + ); + Ok(TeleportAdmitOutcome::Admitted(Box::new(row))) + } + async fn teleport_advance( + &self, + id: TeleportId, + from: TeleportPhase, + to: TeleportPhase, + patch: TeleportPatch, + epoch: i64, + ) -> Result { + self.gate()?; + if !from.can_transition_to(to) { + return Err(MetaError::Conflict(format!( + "illegal teleport phase: {from:?} -> {to:?}" + ))); + } + let mut db = self.db.lock(); + let Some(r) = db.teleports.get(&id) else { + return Ok(false); + }; + if r.phase != from + || db + .sessions + .get(&r.session_id) + .is_none_or(|s| s.current_epoch != epoch) + { + return Ok(false); + } + let r = db.teleports.get_mut(&id).expect("locked teleport"); + r.phase = to; + r.dest_sandbox_id = patch.dest_sandbox_id.or(r.dest_sandbox_id); + r.snapshot_id = patch.snapshot_id.or(r.snapshot_id); + r.export_id = patch.export_id.or(r.export_id.take()); + r.live_payload = patch.live_payload.or(r.live_payload.take()); + r.kind = patch.kind.unwrap_or(r.kind); + r.error = patch.error; + r.attempts += 1; + r.updated_at = self.now(); + r.finished_at = to.is_terminal().then_some(r.updated_at); + let event = append_teleport_finished(&mut db, id, self.now()); + drop(db); + if let Some((sid, idx)) = event { + self.notify( + "session_events", + format!("{{\"session_id\":\"{sid}\",\"idx\":{idx}}}"), + ); + } + Ok(true) + } + async fn teleport_commit(&self, id: TeleportId, epoch: i64) -> Result, MetaError> { + self.gate()?; + let mut db = self.db.lock(); + let Some(r) = db + .teleports + .get(&id) + .cloned() + .filter(|r| r.phase == TeleportPhase::Restored && r.dest_sandbox_id.is_some()) + else { + return Ok(None); + }; + let Some(s) = db.sessions.get_mut(&r.session_id).filter(|s| { + s.current_epoch == epoch + && s.session.status == SessionState::Evacuating + && s.session.host_id == Some(r.source_host_id) + && s.session.sandbox_id == Some(r.source_sandbox_id) + }) else { + return Ok(None); + }; + s.session.host_id = Some(r.dest_host_id); + s.session.sandbox_id = r.dest_sandbox_id; + s.binding_epoch += 1; + s.missing_strikes = 0; + s.updated_at = self.now(); + let minted = s.binding_epoch as u64; + let r = db.teleports.get_mut(&id).expect("locked teleport"); + r.phase = TeleportPhase::Committed; + r.attempts += 1; + r.updated_at = self.now(); + drop(db); + self.notify("placement_changed", "teleport_committed"); + Ok(Some(minted)) + } + async fn teleport_release_source( + &self, + id: TeleportId, + epoch: i64, + how: SourceRelease, + ) -> Result { + self.gate()?; + let mut db = self.db.lock(); + let Some(r) = db.teleports.get(&id).cloned() else { + return Ok(false); + }; + if r.phase != TeleportPhase::Attached + || db + .sessions + .get(&r.session_id) + .is_none_or(|s| s.current_epoch != epoch) + { + return Ok(false); + } + if matches!(how, SourceRelease::SourceHostGone) + && !db + .hosts + .get(&r.source_host_id) + .is_none_or(|h| matches!(h.status, HostStatus::Dead | HostStatus::Retired)) + { + return Ok(false); + } + let now = self.now(); + db.sandbox_tombstones + .entry((r.source_host_id, r.source_sandbox_id)) + .or_insert((Some(r.session_id), now)); + let r = db.teleports.get_mut(&id).expect("locked teleport"); + r.phase = TeleportPhase::Done; + r.attempts += 1; + r.updated_at = now; + r.finished_at = Some(now); + let event = append_teleport_finished(&mut db, id, self.now()); + drop(db); + if let Some((sid, idx)) = event { + self.notify( + "session_events", + format!("{{\"session_id\":\"{sid}\",\"idx\":{idx}}}"), + ); + } + Ok(true) + } + async fn teleport_settle( + &self, + id: TeleportId, + epoch: i64, + settle: TeleportSettle, + ) -> Result { + self.gate()?; + let now = self.now(); + let mut db = self.db.lock(); + let Some(r) = db.teleports.get(&id).cloned() else { + return Ok(false); + }; + if r.phase.is_terminal() + || db + .sessions + .get(&r.session_id) + .is_none_or(|s| s.current_epoch != epoch) + { + return Ok(false); + } + let mut events = Vec::new(); + if let Some(target) = settle.session { + let s = db.sessions.get_mut(&r.session_id).expect("fenced session"); + let current = s.session.status; + current + .try_transition_to(target) + .map_err(|e| MetaError::Conflict(e.to_string()))?; + s.session.status = target; + s.session.sandbox_id = None; + s.updated_at = now; + if let Some(idx) = append_event_idempotent_locked( + &mut db, + r.session_id, + &format!("teleport:{id}:settled"), + "status_changed", + serde_json::json!({"type":"status_changed","from":current.as_str(),"to":target.as_str(),"at":now}), + now, + )? { + events.push((r.session_id, idx)); + } + } + if settle.entomb_source { + db.sandbox_tombstones + .entry((r.source_host_id, r.source_sandbox_id)) + .or_insert((Some(r.session_id), now)); + } + let row = db.teleports.get_mut(&id).expect("locked teleport"); + row.phase = TeleportPhase::Failed; + row.error = Some(settle.error); + row.attempts += 1; + row.updated_at = now; + row.finished_at = Some(now); + events.extend(append_teleport_finished(&mut db, id, now)); + drop(db); + for (sid, idx) in events { + self.notify( + "session_events", + format!("{{\"session_id\":\"{sid}\",\"idx\":{idx}}}"), + ); + } + if settle.session.is_some() { + self.notify("placement_changed", "teleport_settled"); + } + Ok(true) + } + async fn teleport_abort(&self, id: TeleportId, epoch: i64) -> Result { + self.gate()?; + let error = self + .db + .lock() + .teleports + .get(&id) + .and_then(|r| r.error.clone()); + self.teleport_advance( + id, + TeleportPhase::RollingBack, + TeleportPhase::Aborted, + TeleportPatch { + error, + ..Default::default() + }, + epoch, + ) + .await + } + async fn list_open_teleports(&self) -> Result, MetaError> { + self.gate()?; + let mut rows: Vec<_> = self + .db + .lock() + .teleports + .values() + .filter(|r| !r.phase.is_terminal()) + .cloned() + .collect(); + rows.sort_by_key(|r| (r.created_at, r.id)); + Ok(rows) + } + async fn open_teleport_for_session( + &self, + sid: SessionId, + ) -> Result, MetaError> { + self.gate()?; + Ok(self + .db + .lock() + .teleports + .values() + .find(|r| r.session_id == sid && !r.phase.is_terminal()) + .cloned()) + } + // ================= sessions ================= /// `INSERT INTO sessions (id, status='pending', image_uri, mode, @@ -282,7 +679,6 @@ impl MetadataStore for SimMetadataStore { recovery_epoch: 0, shell_pinned_until: None, durable_head: None, - evac_attempts: 0, evict_attempts: 0, updated_at: now, }, @@ -411,7 +807,6 @@ impl MetadataStore for SimMetadataStore { recovery_epoch: 0, shell_pinned_until: None, durable_head: None, - evac_attempts: 0, evict_attempts: 0, updated_at: now, }, @@ -491,9 +886,6 @@ impl MetadataStore for SimMetadataStore { }; row.session.last_active_at = now; row.updated_at = now; - if target == SessionState::Evacuating { - row.evac_attempts = 0; - } if target == SessionState::Evicting { row.evict_attempts = 0; } @@ -1052,8 +1444,15 @@ impl MetadataStore for SimMetadataStore { let mut out = Vec::new(); let mut log = Vec::new(); let mut tombstones = Vec::new(); + let owned: std::collections::BTreeSet<_> = db + .teleports + .values() + .filter(|r| r.source_host_id == host_id && !r.phase.is_terminal()) + .map(|r| r.session_id) + .collect(); for row in db.sessions.values_mut() { if row.session.host_id == Some(host_id) + && !owned.contains(&row.session.id) && !matches!( row.session.status, SessionState::Completed | SessionState::Failed | SessionState::Dead @@ -1124,10 +1523,19 @@ impl MetadataStore for SimMetadataStore { .capture_jobs .values() .any(|job| job.host_id == Some(host_id) && !job.stage.is_terminal()); + // ADR 0123 B: an open move owns both endpoints (see the PG twin). + let move_owned = |s: &engram_core::SandboxId| { + db.teleports.values().any(|t| { + !t.phase.is_terminal() + && (t.source_sandbox_id == *s + || t.dest_sandbox_id == Some(*s) + || (t.dest_host_id == host_id && t.dest_sandbox_id.is_none())) + }) + }; let unbound: Vec = running .iter() .copied() - .filter(|s| !capture_active && !bound.contains(s)) + .filter(|s| !capture_active && !bound.contains(s) && !move_owned(s)) .collect(); db.sandbox_unbound_sightings .retain(|(h, s), _| *h != host_id || unbound.contains(s)); @@ -2046,9 +2454,6 @@ impl MetadataStore for SimMetadataStore { }; row.session.last_active_at = now; row.updated_at = now; - if to == SessionState::Evacuating { - row.evac_attempts = 0; - } if to == SessionState::Evicting { row.evict_attempts = 0; } @@ -2106,9 +2511,6 @@ impl MetadataStore for SimMetadataStore { if matches!(disposition, BindingDisposition::Detach) { row.session.sandbox_id = None; } - if to == SessionState::Evacuating { - row.evac_attempts = 0; - } if to == SessionState::Evicting { row.evict_attempts = 0; } @@ -3348,17 +3750,6 @@ impl MetadataStore for SimMetadataStore { Ok(()) } - async fn list_evacuating_sessions(&self) -> Result, MetaError> { - self.gate()?; - let db = self.db.lock(); - Ok(db - .sessions - .values() - .filter(|r| r.session.status == SessionState::Evacuating) - .map(|r| (r.session.clone(), r.evac_attempts.max(0) as u32)) - .collect()) - } - async fn list_host_lost_sessions(&self) -> Result, MetaError> { self.gate()?; let db = self.db.lock(); @@ -3471,17 +3862,6 @@ impl MetadataStore for SimMetadataStore { Ok(Some(indices)) } - async fn bump_evac_attempts(&self, session_id: SessionId) -> Result { - self.gate()?; - let mut db = self.db.lock(); - let r = db - .sessions - .get_mut(&session_id) - .ok_or(MetaError::NotFound)?; - r.evac_attempts += 1; - Ok(r.evac_attempts.max(0) as u32) - } - async fn bump_evict_attempts(&self, session_id: SessionId) -> Result { self.gate()?; let mut db = self.db.lock(); @@ -4167,14 +4547,6 @@ impl MetadataStore for SimMetadataStore { panic!("SimMeta: get_skill_by_name not implemented — add it plus a conformance case (ADR 0098 D4)") } - async fn get_teleport_target( - &self, - _id: SessionId, - ) -> Result>)>, MetaError> { - self.gate()?; - Ok(self.db.lock().teleport_targets.get(&_id).copied()) - } - async fn hosts_with_live_capture_jobs( &self, ) -> Result, MetaError> { @@ -4904,27 +5276,6 @@ impl MetadataStore for SimMetadataStore { panic!("SimMeta: set_enable_job_state not implemented — add it plus a conformance case (ADR 0098 D4)") } - async fn set_teleport_target( - &self, - id: SessionId, - target: Option, - ) -> Result<(), MetaError> { - // PG twin: set/clear the pin + its `_set_at` together (issue #214). - // A no-op for an absent session, like the bare UPDATE. - self.gate()?; - let now = self.now(); - let mut db = self.db.lock(); - match target { - Some(h) => { - db.teleport_targets.insert(id, (h, Some(now))); - } - None => { - db.teleport_targets.remove(&id); - } - } - Ok(()) - } - /// `SELECT count(*), coalesce(sum(size_bytes),0) FROM snapshots` — /// the fleet view's storage aggregate over ALL snapshot rows (session /// captures AND template/base snapshots). diff --git a/crates/engram-sim/tests/meta_conformance.rs b/crates/engram-sim/tests/meta_conformance.rs index 018ab122a..971402f1c 100644 --- a/crates/engram-sim/tests/meta_conformance.rs +++ b/crates/engram-sim/tests/meta_conformance.rs @@ -3060,21 +3060,6 @@ async fn broker_token_flow(ctx: &Ctx) { .is_none()); } -/// ADR 0045 teleport target pin (R2): set stamps host+`_set_at`, clear -/// nulls both, get round-trips. -async fn teleport_target_flow(ctx: &Ctx) { - let meta = &ctx.meta; - let id = meta.create_session(spec("conf:teleport")).await.unwrap(); - assert!(meta.get_teleport_target(id).await.unwrap().is_none()); - let host = HostId::new(); - meta.set_teleport_target(id, Some(host)).await.unwrap(); - let (got_host, set_at) = meta.get_teleport_target(id).await.unwrap().expect("pinned"); - assert_eq!(got_host, host); - assert!(set_at.is_some(), "a set pin stamps its set_at"); - meta.set_teleport_target(id, None).await.unwrap(); - assert!(meta.get_teleport_target(id).await.unwrap().is_none()); -} - /// ADR 0101 C: the parked lifecycle + the durability-floor settle. /// `parked` is a real state (`list_parked_sessions` finds it, the /// eviction sweep does not), and `settle_evicted_session_idle` is a @@ -4065,7 +4050,6 @@ conformance!( t_parked_lifecycle_and_eviction_settle, super::parked_lifecycle_and_eviction_settle ); -conformance!(t_teleport_target_flow, super::teleport_target_flow); conformance!(t_session_lifecycle, super::session_lifecycle); conformance!( t_binding_disposition_contract, @@ -5516,6 +5500,7 @@ async fn teleport_fixture( cpu_budget_vcpus: 2, snapshot_id: None, export_id: None, + live_payload: None, attempts: 0, error: None, created_at: ctx.clock.now_utc(), @@ -6192,3 +6177,522 @@ conformance!( t_harness_event_delivery_dedups, super::harness_event_delivery_dedups ); + +async fn teleport_admission_fixture( + ctx: &Ctx, +) -> engram_core::types::teleport::TeleportAdmitRequest { + use engram_core::types::teleport::*; + let source = HostId::new(); + let dest = HostId::new(); + for h in [source, dest] { + let mut row = host_record(h, "teleport", ctx.clock.now_utc()); + row.utilization.allocatable_mib = 4096; + ctx.meta.upsert_host(row).await.unwrap(); + let mut hb = heartbeat_fixture(); + hb.utilization.allocatable_mib = 4096; + ctx.meta.touch_host_heartbeat(h, hb).await.unwrap(); + } + let sid = SessionId::new(); + let ws = engram_core::traits::metadata::SessionCreateWriteSet { + session_id: sid, + spec: spec("conf:teleport-machine"), + mem_budget_mib: 4096, + cpu_budget_vcpus: 2, + sealed_secrets: None, + capabilities: Vec::new(), + integration_policy_json: None, + runtime_spec: engram_core::types::runtime_spec::RuntimeSpec::new( + Vec::new(), + None, + None, + Vec::new(), + ), + oauth_binding: None, + }; + assert!(matches!( + ctx.meta + .reserve_and_persist_create(ws, &[source], 0) + .await + .unwrap(), + engram_core::traits::metadata::CreateDisposition::Placed(_) + )); + ctx.meta + .transition_session_created(sid, engram_core::SandboxId::new()) + .await + .unwrap(); + ctx.meta + .transition_session(sid, SessionState::Active, BindingDisposition::Retain) + .await + .unwrap(); + TeleportAdmitRequest { + id: engram_core::TeleportId::new(), + session_id: sid, + reason: TeleportReason::Ui, + epoch: 0, + candidates: vec![source, dest], + pinned_dest: None, + mem_budget_mib: 4096, + cpu_budget_vcpus: 2, + max_open_per_dest: 1, + live_capable: false, + } +} +async fn admit_teleport( + ctx: &Ctx, + req: engram_core::types::teleport::TeleportAdmitRequest, +) -> engram_core::types::teleport::TeleportRow { + let engram_core::types::teleport::TeleportAdmitOutcome::Admitted(row) = + ctx.meta.teleport_admit(req).await.unwrap() + else { + panic!("admission refused") + }; + *row +} +async fn teleport_admit_reserves_dest_under_lock(ctx: &Ctx) { + use engram_core::types::teleport::*; + let req = teleport_admission_fixture(ctx).await; + let sid = req.session_id; + let mut stale = req.clone(); + stale.epoch = 1; + assert!(matches!( + ctx.meta.teleport_admit(stale).await.unwrap(), + TeleportAdmitOutcome::Fenced + )); + let mut too_big = req.clone(); + too_big.mem_budget_mib = 4097; + too_big.pinned_dest = Some(req.candidates[1]); + assert!(matches!( + ctx.meta.teleport_admit(too_big).await.unwrap(), + TeleportAdmitOutcome::NoFit + )); + assert_eq!( + ctx.meta.get_session(sid).await.unwrap().status, + SessionState::Active + ); + assert!(ctx.meta.list_open_teleports().await.unwrap().is_empty()); + let mut zero = req.clone(); + zero.max_open_per_dest = 0; + assert!(matches!( + ctx.meta.teleport_admit(zero).await.unwrap(), + TeleportAdmitOutcome::NoFit + )); + ctx.meta + .set_host_cordon( + req.candidates[1], + Some(engram_core::types::host::CordonOwner::Admin), + None, + ) + .await + .unwrap(); + assert!(matches!( + ctx.meta.teleport_admit(req.clone()).await.unwrap(), + TeleportAdmitOutcome::NoFit + )); + ctx.meta + .set_host_cordon(req.candidates[1], None, None) + .await + .unwrap(); + let row = admit_teleport(ctx, req.clone()).await; + assert_eq!( + ctx.meta.get_session(sid).await.unwrap().host_id, + Some(row.source_host_id) + ); + assert_eq!( + ctx.meta.per_host_reserved().await.unwrap()[&row.dest_host_id].mem_mib, + 4096 + ); + assert_eq!( + ctx.meta.per_host_reserved().await.unwrap()[&row.source_host_id].mem_mib, + 4096 + ); + assert!(matches!( + ctx.meta.teleport_admit(req).await, + Err(MetaError::Conflict(_)) + )); + let mut second = teleport_admission_fixture(ctx).await; + second.candidates = vec![row.dest_host_id]; + assert!(matches!( + ctx.meta.teleport_admit(second.clone()).await.unwrap(), + TeleportAdmitOutcome::NoFit + )); + let mut hb = heartbeat_fixture(); + hb.utilization.allocatable_mib = 32768; + ctx.meta + .touch_host_heartbeat(row.dest_host_id, hb) + .await + .unwrap(); + // RAM now fits both moves; the open-count limit alone refuses this one. + assert!(matches!( + ctx.meta.teleport_admit(second.clone()).await.unwrap(), + TeleportAdmitOutcome::NoFit + )); + ctx.meta + .transition_session( + second.session_id, + SessionState::Evicting, + BindingDisposition::Retain, + ) + .await + .unwrap(); + assert!(matches!( + ctx.meta.teleport_admit(second).await.unwrap(), + TeleportAdmitOutcome::SessionNotActive(SessionState::Evicting) + )); +} +async fn teleport_phase_cas_is_fenced_and_legal(ctx: &Ctx) { + use engram_core::types::teleport::*; + let row = admit_teleport(ctx, teleport_admission_fixture(ctx).await).await; + assert!(!ctx + .meta + .teleport_advance( + row.id, + TeleportPhase::Admitted, + TeleportPhase::Captured, + TeleportPatch::default(), + 1 + ) + .await + .unwrap()); + assert!(matches!( + ctx.meta + .teleport_advance( + row.id, + TeleportPhase::Admitted, + TeleportPhase::Done, + TeleportPatch::default(), + 0 + ) + .await, + Err(MetaError::Conflict(_)) + )); + let patch = TeleportPatch { + export_id: Some("export".into()), + live_payload: Some(serde_json::json!({"peer_token":"test-token"})), + kind: Some(TeleportKind::Snapshot), + ..Default::default() + }; + assert!(ctx + .meta + .teleport_advance( + row.id, + TeleportPhase::Admitted, + TeleportPhase::Admitted, + patch, + 0 + ) + .await + .unwrap()); + assert_eq!( + ctx.meta + .open_teleport_for_session(row.session_id) + .await + .unwrap() + .unwrap() + .export_id + .as_deref(), + Some("export") + ); + assert!(ctx + .meta + .teleport_advance( + row.id, + TeleportPhase::Admitted, + TeleportPhase::Captured, + TeleportPatch::default(), + 0 + ) + .await + .unwrap()); + assert!(!ctx + .meta + .teleport_advance( + row.id, + TeleportPhase::Admitted, + TeleportPhase::Captured, + TeleportPatch::default(), + 0 + ) + .await + .unwrap()); + assert_eq!( + ctx.meta + .open_teleport_for_session(row.session_id) + .await + .unwrap() + .unwrap() + .live_payload, + Some(serde_json::json!({"peer_token":"test-token"})) + ); +} +async fn restored_teleport(ctx: &Ctx) -> engram_core::types::teleport::TeleportRow { + use engram_core::types::teleport::*; + let row = admit_teleport(ctx, teleport_admission_fixture(ctx).await).await; + ctx.meta + .teleport_advance( + row.id, + TeleportPhase::Admitted, + TeleportPhase::Captured, + TeleportPatch::default(), + 0, + ) + .await + .unwrap(); + ctx.meta + .teleport_advance( + row.id, + TeleportPhase::Captured, + TeleportPhase::Restored, + TeleportPatch { + dest_sandbox_id: Some(engram_core::SandboxId::new()), + ..Default::default() + }, + 0, + ) + .await + .unwrap(); + ctx.meta + .open_teleport_for_session(row.session_id) + .await + .unwrap() + .unwrap() +} +async fn teleport_commit_is_one_statement(ctx: &Ctx) { + use engram_core::types::teleport::*; + let row = restored_teleport(ctx).await; + let before = ctx.meta.per_host_reserved().await.unwrap(); + assert_eq!(before[&row.source_host_id].mem_mib, 4096); + assert_eq!(before[&row.dest_host_id].mem_mib, 4096); + assert_eq!(ctx.meta.teleport_commit(row.id, 1).await.unwrap(), None); + assert_eq!( + ctx.meta + .session_binding_generations(row.session_id) + .await + .unwrap(), + (1, 0) + ); + assert_eq!(ctx.meta.teleport_commit(row.id, 0).await.unwrap(), Some(2)); + assert_eq!( + ctx.meta + .session_binding_generations(row.session_id) + .await + .unwrap(), + (2, 0) + ); + assert_eq!(ctx.meta.teleport_commit(row.id, 0).await.unwrap(), None); + let s = ctx.meta.get_session(row.session_id).await.unwrap(); + assert_eq!(s.host_id, Some(row.dest_host_id)); + assert_eq!(s.sandbox_id, row.dest_sandbox_id); + assert_eq!( + ctx.meta + .open_teleport_for_session(row.session_id) + .await + .unwrap() + .unwrap() + .phase, + TeleportPhase::Committed + ); + assert!(ctx + .meta + .sandbox_tombstones_for_host(row.source_host_id) + .await + .unwrap() + .is_empty()); + // The session's own reservation moved to the destination; the source + // keeps the move's budget until release (its VM is still paused there). + let after = ctx.meta.per_host_reserved().await.unwrap(); + assert_eq!( + after.get(&row.source_host_id).map_or(0, |r| r.mem_mib), + row.mem_budget_mib + ); + assert_eq!( + after[&row.dest_host_id].mem_mib, + before[&row.dest_host_id].mem_mib + ); + let mismatch = restored_teleport(ctx).await; + ctx.meta + .assign_session_sandbox(mismatch.session_id, Some(engram_core::SandboxId::new())) + .await + .unwrap(); + assert_eq!( + ctx.meta.teleport_commit(mismatch.id, 0).await.unwrap(), + None + ); +} +async fn teleport_release_entombs_source(ctx: &Ctx) { + use engram_core::types::teleport::*; + let row = restored_teleport(ctx).await; + ctx.meta.teleport_commit(row.id, 0).await.unwrap(); + ctx.meta + .teleport_advance( + row.id, + TeleportPhase::Committed, + TeleportPhase::Attached, + TeleportPatch::default(), + 0, + ) + .await + .unwrap(); + assert!(!ctx + .meta + .teleport_release_source(row.id, 0, SourceRelease::SourceHostGone) + .await + .unwrap()); + assert!(!ctx + .meta + .teleport_release_source(row.id, 1, SourceRelease::DestroyAcked) + .await + .unwrap()); + assert!(ctx + .meta + .teleport_release_source(row.id, 0, SourceRelease::DestroyAcked) + .await + .unwrap()); + assert!(!ctx + .meta + .teleport_release_source(row.id, 0, SourceRelease::DestroyAcked) + .await + .unwrap()); + assert_eq!( + ctx.meta + .sandbox_tombstones_for_host(row.source_host_id) + .await + .unwrap(), + vec![row.source_sandbox_id] + ); + let events = ctx + .meta + .list_session_events_since(row.session_id, -1, 100) + .await + .unwrap(); + let finished: Vec<_> = events + .iter() + .filter(|e| e.kind == "teleport_finished") + .collect(); + assert_eq!(finished.len(), 1); + assert_eq!(finished[0].payload["outcome"], "done"); + assert!(ctx + .meta + .open_teleport_for_session(row.session_id) + .await + .unwrap() + .is_none()); +} +async fn teleport_rollback_keeps_dest_reserved_until_aborted(ctx: &Ctx) { + use engram_core::types::teleport::*; + let row = restored_teleport(ctx).await; + ctx.meta + .teleport_advance( + row.id, + TeleportPhase::Restored, + TeleportPhase::RollingBack, + TeleportPatch { + error: Some("capture_failed".into()), + ..Default::default() + }, + 0, + ) + .await + .unwrap(); + assert_eq!( + ctx.meta.per_host_reserved().await.unwrap()[&row.dest_host_id].mem_mib, + 4096 + ); + assert!(ctx.meta.teleport_abort(row.id, 0).await.unwrap()); + assert!(!ctx + .meta + .per_host_reserved() + .await + .unwrap() + .contains_key(&row.dest_host_id)); + let events = ctx + .meta + .list_session_events_since(row.session_id, -1, 100) + .await + .unwrap(); + let finished: Vec<_> = events + .iter() + .filter(|e| e.kind == "teleport_finished") + .collect(); + assert_eq!(finished.len(), 1); + assert_eq!(finished[0].payload["outcome"], "aborted"); + assert_eq!(finished[0].payload["error"], "capture_failed"); +} +async fn dead_host_orphan_skips_open_teleport_sources(ctx: &Ctx) { + let row = restored_teleport(ctx).await; + assert!(ctx + .meta + .mark_host_dead_if_lease_expired(row.source_host_id) + .await + .unwrap() + .is_empty()); + let s = ctx.meta.get_session(row.session_id).await.unwrap(); + assert_eq!(s.status, SessionState::Evacuating); + assert_eq!(s.sandbox_id, Some(row.source_sandbox_id)); +} +fn lost(error: &str) -> engram_core::types::teleport::TeleportSettle { + engram_core::types::teleport::TeleportSettle { + error: error.into(), + session: None, + entomb_source: false, + } +} +async fn teleport_fail_and_abort_are_fenced(ctx: &Ctx) { + use engram_core::types::teleport::*; + let row = restored_teleport(ctx).await; + assert!(!ctx + .meta + .teleport_settle(row.id, 1, lost("lost")) + .await + .unwrap()); + assert!(!ctx.meta.teleport_abort(row.id, 0).await.unwrap()); + ctx.meta + .teleport_advance( + row.id, + TeleportPhase::Restored, + TeleportPhase::RollingBack, + TeleportPatch::default(), + 0, + ) + .await + .unwrap(); + assert!(!ctx.meta.teleport_abort(row.id, 1).await.unwrap()); + assert!(ctx + .meta + .teleport_settle(row.id, 0, lost("lost")) + .await + .unwrap()); + assert!(!ctx + .meta + .teleport_settle(row.id, 0, lost("again")) + .await + .unwrap()); + assert!(ctx.meta.list_open_teleports().await.unwrap().is_empty()); +} +conformance!( + teleport_admit_reserves_dest_under_lock_test, + teleport_admit_reserves_dest_under_lock +); +conformance!( + teleport_phase_cas_is_fenced_and_legal_test, + teleport_phase_cas_is_fenced_and_legal +); +conformance!( + teleport_commit_is_one_statement_test, + teleport_commit_is_one_statement +); +conformance!( + teleport_release_entombs_source_test, + teleport_release_entombs_source +); +conformance!( + teleport_rollback_keeps_dest_reserved_until_aborted_test, + teleport_rollback_keeps_dest_reserved_until_aborted +); +conformance!( + dead_host_orphan_skips_open_teleport_sources_test, + dead_host_orphan_skips_open_teleport_sources +); +conformance!( + teleport_fail_and_abort_are_fenced_test, + teleport_fail_and_abort_are_fenced +); diff --git a/deploy/migrations/0122_teleport_machine.sql b/deploy/migrations/0122_teleport_machine.sql new file mode 100644 index 000000000..685d671a1 --- /dev/null +++ b/deploy/migrations/0122_teleport_machine.sql @@ -0,0 +1,44 @@ +ALTER TABLE sessions + DROP COLUMN evac_attempts, + DROP COLUMN teleport_target_host_id, + DROP COLUMN teleport_target_set_at; +DROP INDEX IF EXISTS idx_sessions_evacuating; + +-- A successor restores from the same export without repeating presetup. +ALTER TABLE session_teleports ADD COLUMN live_payload JSONB; + +-- A session mid-move under the retired evacuation scanner has no +-- session_teleports row, and from this version nothing else drives +-- `evacuating`. Settle those rows the way a lost host settles (the +-- dead-host orphan step): entomb the bound sandbox, drop the binding, and +-- keep the live disk manifest. Idle when a recoverable memory snapshot OR a +-- live disk manifest exists (the latter resumes through the disk-only cold +-- boot); Dead only with nothing recoverable. Rows with an open teleport +-- belong to the new machine. +INSERT INTO sandbox_tombstones (host_id, sandbox_id, session_id, created_at) +SELECT host_id, sandbox_id, id, now() + FROM sessions + WHERE status = 'evacuating' + AND host_id IS NOT NULL + AND sandbox_id IS NOT NULL + AND NOT EXISTS ( + SELECT 1 FROM session_teleports t + WHERE t.session_id = sessions.id + AND t.phase NOT IN ('done', 'aborted', 'failed')) +ON CONFLICT DO NOTHING; + +UPDATE sessions + SET status = CASE + WHEN live_disk_manifest_id IS NOT NULL + OR EXISTS (SELECT 1 FROM snapshots s + WHERE s.session_id = sessions.id AND s.recoverable) + THEN 'idle' ELSE 'dead' + END, + host_id = NULL, + sandbox_id = NULL, + updated_at = now() + WHERE status = 'evacuating' + AND NOT EXISTS ( + SELECT 1 FROM session_teleports t + WHERE t.session_id = sessions.id + AND t.phase NOT IN ('done', 'aborted', 'failed')); diff --git a/docs/adr/0123-durable-teleport-and-host-retirement.md b/docs/adr/0123-durable-teleport-and-host-retirement.md index a0d3cc205..8a1eac09a 100644 --- a/docs/adr/0123-durable-teleport-and-host-retirement.md +++ b/docs/adr/0123-durable-teleport-and-host-retirement.md @@ -560,3 +560,49 @@ pub enum HarnessFrame { finished destination drain is remembered so a failed role persist is retried by the next `migration_drain_wait` rather than reported as a lost drain. +- 2026-10-06 (adversarial review, coordinator): every teleport write + locks the session row under its `current_epoch` FIRST (`lock_teleport_ + session_at_epoch`), so a stale driver cannot land a phase change after a + reclaim; a failed move settles in ONE fenced transaction + (`teleport_settle`: session terminal state, source tombstone, failed row, + both events) and `teleport_fail` is gone. The SOURCE keeps the move's + budget in `committed|attached` (`source_reserving_phases`) because its VM + still occupies that RAM until release. The heartbeat's stably-unbound + cleanup never entombs a sandbox an open move names, nor any unbound + sandbox on a host an open move is restoring into. The host's ownership + question answers `owned` for both endpoints of an open move, so the + export TTL sweep never destroys a post-commit source under a draining + destination. A deleted host row counts as gone. A consumed export on + commit (`NotFound`) proceeds to the destroy ack. Deterministic + `HarnessSpawn` failures share ordinary resume's classifier. +- 2026-10-06 (adversarial review, host): `snapshot_hold` persists a + `held-source` manifest role before the pause, so a host-agent restart + re-adopts the paused VM as a frozen source (never a checkpoint + candidate, never self-resumed); `resume` clears it. The detached swap + re-arm runs under the capture lock. The abort's cleanup after the export + is consumed runs in a detached task so a cancelled request cannot leave + the fence raised. A failed post-restore dirty-file relocation destroys + the VM before its NBD state is dropped. +- 2026-10-06 (adversarial review, harness wire): `SeqEvent` carries a + per-process `incarnation`, and the delivery key is + `harness:{epoch}:{incarnation}:{seq}`, so a fresh process at the same + binding epoch never collides with its predecessor's keys. On a + connection at a HIGHER epoch the SDK announces `RunContinued` for every + run it has sequenced a start for and no end, BEFORE any event the engine + buffered across the cut; the engines no longer announce it themselves. + The outbox acknowledgement for a confirming event runs on a replayed + duplicate too and a failed acknowledgement withholds the harness ack. + The command forwarder never awaits the engine's channel (a full channel + drops the connection; durable prompts are redelivered). Known limits: a + legacy `Event` frame (an old memory-resident harness) does not advance + readiness and its POST retry is not idempotent; such a process exits on + `Superseded` and the respawn uses the staged (new) harness. +- 2026-10-06 (adversarial review, operator): every live victim asks + `RetireHost` before any release decision (a grant that landed before a + failed `removing` patch is still observed); `FailedPrecondition` is a + foreign cordon and releases only the annotation; `mark_victim` adds a + Node finalizer so the record outlives the cloud instance, and + `DeleteHost` runs only once the Node object is being deleted; a Node + that was unschedulable before it became a victim keeps that cordon; the + GKE actuator refuses pools backed by more than one instance group + (setSize is per zone). diff --git a/orchestrator/src/authz/policy-map.ts b/orchestrator/src/authz/policy-map.ts index 98dcadbf2..e85d74856 100644 --- a/orchestrator/src/authz/policy-map.ts +++ b/orchestrator/src/authz/policy-map.ts @@ -198,7 +198,7 @@ export const POLICY: Record = { "FleetService.DeleteHost": { action: "manage", subject: "all" }, "FleetService.GetStorageSummary": { action: "manage", subject: "all" }, "FleetService.FlushSession": { action: "manage", subject: "all" }, - "FleetService.EvacuateSession": { action: "manage", subject: "all" }, + "FleetService.TeleportSession": { action: "manage", subject: "all" }, "FleetService.ChunkGc": { action: "manage", subject: "all" }, "FleetService.BundleGc": { action: "manage", subject: "all" }, "FleetService.SnapshotBlobGc": { action: "manage", subject: "all" }, diff --git a/orchestrator/src/control-plane/session-events.ts b/orchestrator/src/control-plane/session-events.ts index 70aac448f..329595302 100644 --- a/orchestrator/src/control-plane/session-events.ts +++ b/orchestrator/src/control-plane/session-events.ts @@ -23,6 +23,7 @@ export const CURATED_KINDS: ReadonlySet = new Set([ "run_completed", "run_interrupted", "run_continued", + "teleport_finished", "harness_idle", "harness_parked", "resumed", diff --git a/orchestrator/src/gen/engram/app/v1/fleet_pb.ts b/orchestrator/src/gen/engram/app/v1/fleet_pb.ts index 7c7e01ea0..7da919176 100644 --- a/orchestrator/src/gen/engram/app/v1/fleet_pb.ts +++ b/orchestrator/src/gen/engram/app/v1/fleet_pb.ts @@ -12,7 +12,7 @@ import type { Message } from "@bufbuild/protobuf"; * Describes the file engram/app/v1/fleet.proto. */ export const file_engram_app_v1_fleet: GenFile = /*@__PURE__*/ - fileDesc("ChllbmdyYW0vYXBwL3YxL2ZsZWV0LnByb3RvEg1lbmdyYW0uYXBwLnYxIhIKEExpc3RIb3N0c1JlcXVlc3QiOwoRTGlzdEhvc3RzUmVzcG9uc2USJgoFaG9zdHMYASADKAsyFy5lbmdyYW0uYXBwLnYxLkhvc3RWaWV3IvEGCghIb3N0VmlldxIKCgJpZBgBIAEoCRIQCghob3N0bmFtZRgCIAEoCRIOCgZzdGF0dXMYAyABKAkSGgoSY2FwYWNpdHlfdG90YWxfbWliGAQgASgEEhkKEWNhcGFjaXR5X3VzZWRfbWliGAUgASgEEhkKEXJ1bm5pbmdfc2FuZGJveGVzGAYgASgNEhQKDHJlYWR5X2ltYWdlcxgIIAEoBBIbChNyZWFkeV9pbWFnZV9kaWdlc3RzGAkgAygJEhsKE3V0aWxfZGlza190b3RhbF9taWIYCiABKAQSGgoSdXRpbF9kaXNrX3VzZWRfbWliGAsgASgEEhoKEnV0aWxfbWVtX3RvdGFsX21pYhgMIAEoBBIZChF1dGlsX21lbV91c2VkX21pYhgNIAEoBBIUCgx1dGlsX2NwdV9wY3QYDiABKAISGQoRbGFzdF9oZWFydGJlYXRfYXQYDyABKAkSEAoIY29yZG9uZWQYECABKAgSFwoPYWxsb2NhdGFibGVfbWliGBEgASgEEhQKDHJlc2VydmVkX21pYhgSIAEoBBIQCghmcmVlX21pYhgTIAEoBBITCgt0b3RhbF92Y3B1cxgUIAEoDRIYChBjcHVfYnVkZ2V0X3ZjcHVzGBUgASgEEhYKDnJlc2VydmVkX3ZjcHVzGBYgASgEEhIKCmZyZWVfdmNwdXMYFyABKAQSGQoRdXRpbF9iYXNlX3NobV9taWIYGCABKAQSGwoTdXRpbF9wYXJrZWRfcHNzX21pYhgZIAEoBBIcChR1dGlsX3J1bm5pbmdfcHNzX21pYhgaIAEoBBIcChRmYWlsaW5nX2NhcGFiaWxpdGllcxgbIAMoCRIbChNmY19zbmFwc2hvdF92ZXJzaW9uGBwgASgJEhsKE2NhcGFiaWxpdGllc19zY2hlbWEYHSABKA0SGQoRbGl2ZV9tYXRlcmlhbGl6ZXMYHiABKA0SGQoRbGl2ZV9jYXB0dXJlX2pvYnMYHyABKA0SHwoXdXRpbF9jb21taXR0ZWRfc3dhcF9taWIYICABKAQSFAoMY29yZG9uX293bmVyGCEgASgJEjEKCnJldGlyZW1lbnQYIiABKAsyHS5lbmdyYW0uYXBwLnYxLkhvc3RSZXRpcmVtZW50SgQIBxAIUg9sb2NhbF9zbmFwc2hvdHMiIQoOR2V0SG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCSI4Cg9HZXRIb3N0UmVzcG9uc2USJQoEaG9zdBgBIAEoCzIXLmVuZ3JhbS5hcHAudjEuSG9zdFZpZXciKQoWR2V0SG9zdENvd1N0YXRlUmVxdWVzdBIPCgdob3N0X2lkGAEgASgJIlkKF0dldEhvc3RDb3dTdGF0ZVJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSLQoIc2Vzc2lvbnMYAiADKAsyGy5lbmdyYW0uYXBwLnYxLkNvd1N0YXRlVmlldyIjChBEcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkiEwoRRHJhaW5Ib3N0UmVzcG9uc2UiKAoVQWRtaW5EcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkibAoWQWRtaW5EcmFpbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEhIKCmV2YWN1YXRpbmcYAiADKAkSLQoIZmFpbHVyZXMYAyADKAsyGy5lbmdyYW0uYXBwLnYxLkRyYWluRmFpbHVyZSIxCgxEcmFpbkZhaWx1cmUSEgoKc2Vzc2lvbl9pZBgBIAEoCRINCgVlcnJvchgCIAEoCSIzChFDb3Jkb25Ib3N0UmVxdWVzdBIPCgdob3N0X2lkGAEgASgJEg0KBW93bmVyGAIgASgJIjwKF0JlZ2luSG9zdEhhbmRvZmZSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkSEAoIdHRsX3NlY3MYAiABKAQiPQoYQmVnaW5Ib3N0SGFuZG9mZlJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSEAoIYWNjZXB0ZWQYAiABKAgiNQoSQ29yZG9uSG9zdFJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSDgoGc3RhdHVzGAIgASgJIjUKE1VuY29yZG9uSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCRINCgVvd25lchgCIAEoCSI3ChRVbmNvcmRvbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEg4KBnN0YXR1cxgCIAEoCSIkChFEZWxldGVIb3N0UmVxdWVzdBIPCgdob3N0X2lkGAEgASgJIhQKEkRlbGV0ZUhvc3RSZXNwb25zZSIaChhHZXRTdG9yYWdlU3VtbWFyeVJlcXVlc3QihAIKGUdldFN0b3JhZ2VTdW1tYXJ5UmVzcG9uc2USEQoJc25hcHNob3RzGAEgASgEEhYKDnNuYXBzaG90X2J5dGVzGAIgASgEEhIKCmdjX3BlbmRpbmcYAyABKAQSGQoRdHJhY2tlZF9zYW5kYm94ZXMYBCABKAQSFAoMZGlydHlfY2h1bmtzGAUgASgEEhcKD3VuZmx1c2hlZF9ieXRlcxgGIAEoBBIYChBhdmdfbG9jYWxpdHlfcGN0GAcgASgNEioKBHJvd3MYCCADKAsyHC5lbmdyYW0uYXBwLnYxLkR1cmFiaWxpdHlSb3cSGAoQZ2NfcGVuZGluZ19leGFjdBgJIAEoCCLlAQoNRHVyYWJpbGl0eVJvdxISCgpzYW5kYm94X2lkGAEgASgJEhcKCnNlc3Npb25faWQYAiABKAlIAIgBARIPCgdob3N0X2lkGAMgASgJEhQKDGRpcnR5X2NodW5rcxgEIAEoDRITCgtkaXJ0eV9ieXRlcxgFIAEoBBITCgtiYXNlX2NodW5rcxgGIAEoDRIZChFiYXNlX2NodW5rc19sb2NhbBgHIAEoDRIaCg1sYXN0X2ZsdXNoX2F0GAggASgJSAGIAQFCDQoLX3Nlc3Npb25faWRCEAoOX2xhc3RfZmx1c2hfYXQiKQoTRmx1c2hTZXNzaW9uUmVxdWVzdBISCgpzZXNzaW9uX2lkGAEgASgJIlsKFEZsdXNoU2Vzc2lvblJlc3BvbnNlEg8KB291dGNvbWUYASABKAkSHQoQbWFuaWZlc3RfdmVyc2lvbhgCIAEoBEgAiAEBQhMKEV9tYW5pZmVzdF92ZXJzaW9uIlYKFkV2YWN1YXRlU2Vzc2lvblJlcXVlc3QSEgoKc2Vzc2lvbl9pZBgBIAEoCRIYCgt0YXJnZXRfaG9zdBgCIAEoCUgAiAEBQg4KDF90YXJnZXRfaG9zdCI9ChdFdmFjdWF0ZVNlc3Npb25SZXNwb25zZRISCgpzZXNzaW9uX2lkGAEgASgJEg4KBnN0YXR1cxgCIAEoCSIXChVHZXRGbGVldERlbWFuZFJlcXVlc3Qi8QEKFkdldEZsZWV0RGVtYW5kUmVzcG9uc2USEwoLcmVhZHlfaG9zdHMYASABKA0SGQoRc2NoZWR1bGFibGVfaG9zdHMYAiABKA0SEAoIZnJlZV9taWIYAyABKAQSEQoJdG90YWxfbWliGAQgASgEEhIKCmZyZWVfdmNwdXMYBSABKAQSEwoLdG90YWxfdmNwdXMYBiABKAQSFgoOY29yZG9uZWRfaG9zdHMYByABKA0SFwoPcXVldWVkX3Nlc3Npb25zGAggASgEEhIKCnF1ZXVlZF9taWIYCSABKAQSFAoMcXVldWVkX3ZjcHVzGAogASgEIkkKDkNodW5rR2NSZXF1ZXN0Eg8KB2RyeV9ydW4YASABKAgSFwoKZ3JhY2Vfc2VjcxgCIAEoBEgAiAEBQg0KC19ncmFjZV9zZWNzIowCCg9DaHVua0djUmVzcG9uc2USFQoNbGlzdGVkX2NodW5rcxgBIAEoBBIWCg5tYWxmb3JtZWRfa2V5cxgCIAEoBBIUCgxwaW5fc2V0X3NpemUYAyABKAQSGQoRY2FuZGlkYXRlc19tYXJrZWQYBCABKAQSGAoQcHJvbW90ZWRfZGVsZXRlcxgHIAEoBBIdChVwcm9tb3RlX2RlbGV0ZV9lcnJvcnMYCCABKAQSEgoKZ3JhY2Vfc2VjcxgJIAEoBBIYChBnZW5lcmF0aW9uX21vdmVkGAogASgIEhcKCm1hcmtfZXJyb3IYCyABKAlIAIgBAUINCgtfbWFya19lcnJvckoECAUQBkoECAYQByJKCg9CdW5kbGVHY1JlcXVlc3QSDwoHZHJ5X3J1bhgBIAEoCBIXCgpncmFjZV9zZWNzGAIgASgESACIAQFCDQoLX2dyYWNlX3NlY3MiwwEKEEJ1bmRsZUdjUmVzcG9uc2USDgoGbGlzdGVkGAEgASgEEhQKDHBpbl9zZXRfc2l6ZRgCIAEoBBIZChFjYW5kaWRhdGVzX21hcmtlZBgDIAEoBBIYChBwcm9tb3RlZF9kZWxldGVzGAQgASgEEh0KFXByb21vdGVfZGVsZXRlX2Vycm9ycxgFIAEoBBIVCg1yZXN0YXJ0X2NvdW50GAYgASgNEh4KFnByb21vdGVfcmVwaW5uZWRfc2tpcHMYByABKAQiUAoVU25hcHNob3RCbG9iR2NSZXF1ZXN0Eg8KB2RyeV9ydW4YASABKAgSFwoKZ3JhY2Vfc2VjcxgCIAEoBEgAiAEBQg0KC19ncmFjZV9zZWNzItwBChZTbmFwc2hvdEJsb2JHY1Jlc3BvbnNlEg4KBmxpc3RlZBgBIAEoBBIRCgltYWxmb3JtZWQYAiABKAQSFAoMcGluX3NldF9zaXplGAMgASgEEhkKEWNhbmRpZGF0ZXNfbWFya2VkGAQgASgEEhgKEHByb21vdGVkX2RlbGV0ZXMYBSABKAQSHgoWcHJvbW90ZV9yZXBpbm5lZF9za2lwcxgGIAEoBBIdChVwcm9tb3RlX2RlbGV0ZV9lcnJvcnMYByABKAQSFQoNcmVzdGFydF9jb3VudBgIIAEoDSJDChFSZXRpcmVIb3N0UmVxdWVzdBIPCgdob3N0X2lkGAEgASgJEg0KBW93bmVyGAIgASgJEg4KBnJlYXNvbhgDIAEoCSJzChJSZXRpcmVIb3N0UmVzcG9uc2USDwoHaG9zdF9pZBgBIAEoCRIxCgpyZXRpcmVtZW50GAIgASgLMh0uZW5ncmFtLmFwcC52MS5Ib3N0UmV0aXJlbWVudBIZChF0ZWxlcG9ydHNfcGxhbm5lZBgDIAEoDSJuCg5Ib3N0UmV0aXJlbWVudBIUCgxyZXF1ZXN0ZWRfYXQYASABKAkSEgoKcmV0aXJlZF9hdBgCIAEoCRIyCghibG9ja2VycxgDIAMoCzIgLmVuZ3JhbS5hcHAudjEuUmV0aXJlbWVudEJsb2NrZXIiMAoRUmV0aXJlbWVudEJsb2NrZXISDAoEa2luZBgBIAEoCRINCgVjb3VudBgCIAEoBDLoCwoMRmxlZXRTZXJ2aWNlEk4KCUxpc3RIb3N0cxIfLmVuZ3JhbS5hcHAudjEuTGlzdEhvc3RzUmVxdWVzdBogLmVuZ3JhbS5hcHAudjEuTGlzdEhvc3RzUmVzcG9uc2USSAoHR2V0SG9zdBIdLmVuZ3JhbS5hcHAudjEuR2V0SG9zdFJlcXVlc3QaHi5lbmdyYW0uYXBwLnYxLkdldEhvc3RSZXNwb25zZRJRCgpSZXRpcmVIb3N0EiAuZW5ncmFtLmFwcC52MS5SZXRpcmVIb3N0UmVxdWVzdBohLmVuZ3JhbS5hcHAudjEuUmV0aXJlSG9zdFJlc3BvbnNlEmAKD0dldEhvc3RDb3dTdGF0ZRIlLmVuZ3JhbS5hcHAudjEuR2V0SG9zdENvd1N0YXRlUmVxdWVzdBomLmVuZ3JhbS5hcHAudjEuR2V0SG9zdENvd1N0YXRlUmVzcG9uc2USTgoJRHJhaW5Ib3N0Eh8uZW5ncmFtLmFwcC52MS5EcmFpbkhvc3RSZXF1ZXN0GiAuZW5ncmFtLmFwcC52MS5EcmFpbkhvc3RSZXNwb25zZRJdCg5BZG1pbkRyYWluSG9zdBIkLmVuZ3JhbS5hcHAudjEuQWRtaW5EcmFpbkhvc3RSZXF1ZXN0GiUuZW5ncmFtLmFwcC52MS5BZG1pbkRyYWluSG9zdFJlc3BvbnNlElEKCkNvcmRvbkhvc3QSIC5lbmdyYW0uYXBwLnYxLkNvcmRvbkhvc3RSZXF1ZXN0GiEuZW5ncmFtLmFwcC52MS5Db3Jkb25Ib3N0UmVzcG9uc2USYwoQQmVnaW5Ib3N0SGFuZG9mZhImLmVuZ3JhbS5hcHAudjEuQmVnaW5Ib3N0SGFuZG9mZlJlcXVlc3QaJy5lbmdyYW0uYXBwLnYxLkJlZ2luSG9zdEhhbmRvZmZSZXNwb25zZRJXCgxVbmNvcmRvbkhvc3QSIi5lbmdyYW0uYXBwLnYxLlVuY29yZG9uSG9zdFJlcXVlc3QaIy5lbmdyYW0uYXBwLnYxLlVuY29yZG9uSG9zdFJlc3BvbnNlElEKCkRlbGV0ZUhvc3QSIC5lbmdyYW0uYXBwLnYxLkRlbGV0ZUhvc3RSZXF1ZXN0GiEuZW5ncmFtLmFwcC52MS5EZWxldGVIb3N0UmVzcG9uc2USZgoRR2V0U3RvcmFnZVN1bW1hcnkSJy5lbmdyYW0uYXBwLnYxLkdldFN0b3JhZ2VTdW1tYXJ5UmVxdWVzdBooLmVuZ3JhbS5hcHAudjEuR2V0U3RvcmFnZVN1bW1hcnlSZXNwb25zZRJXCgxGbHVzaFNlc3Npb24SIi5lbmdyYW0uYXBwLnYxLkZsdXNoU2Vzc2lvblJlcXVlc3QaIy5lbmdyYW0uYXBwLnYxLkZsdXNoU2Vzc2lvblJlc3BvbnNlEmAKD0V2YWN1YXRlU2Vzc2lvbhIlLmVuZ3JhbS5hcHAudjEuRXZhY3VhdGVTZXNzaW9uUmVxdWVzdBomLmVuZ3JhbS5hcHAudjEuRXZhY3VhdGVTZXNzaW9uUmVzcG9uc2USSAoHQ2h1bmtHYxIdLmVuZ3JhbS5hcHAudjEuQ2h1bmtHY1JlcXVlc3QaHi5lbmdyYW0uYXBwLnYxLkNodW5rR2NSZXNwb25zZRJLCghCdW5kbGVHYxIeLmVuZ3JhbS5hcHAudjEuQnVuZGxlR2NSZXF1ZXN0Gh8uZW5ncmFtLmFwcC52MS5CdW5kbGVHY1Jlc3BvbnNlEl0KDlNuYXBzaG90QmxvYkdjEiQuZW5ncmFtLmFwcC52MS5TbmFwc2hvdEJsb2JHY1JlcXVlc3QaJS5lbmdyYW0uYXBwLnYxLlNuYXBzaG90QmxvYkdjUmVzcG9uc2USXQoOR2V0RmxlZXREZW1hbmQSJC5lbmdyYW0uYXBwLnYxLkdldEZsZWV0RGVtYW5kUmVxdWVzdBolLmVuZ3JhbS5hcHAudjEuR2V0RmxlZXREZW1hbmRSZXNwb25zZWIGcHJvdG8z", [file_engram_app_v1_session]); + fileDesc("ChllbmdyYW0vYXBwL3YxL2ZsZWV0LnByb3RvEg1lbmdyYW0uYXBwLnYxIhIKEExpc3RIb3N0c1JlcXVlc3QiOwoRTGlzdEhvc3RzUmVzcG9uc2USJgoFaG9zdHMYASADKAsyFy5lbmdyYW0uYXBwLnYxLkhvc3RWaWV3IvEGCghIb3N0VmlldxIKCgJpZBgBIAEoCRIQCghob3N0bmFtZRgCIAEoCRIOCgZzdGF0dXMYAyABKAkSGgoSY2FwYWNpdHlfdG90YWxfbWliGAQgASgEEhkKEWNhcGFjaXR5X3VzZWRfbWliGAUgASgEEhkKEXJ1bm5pbmdfc2FuZGJveGVzGAYgASgNEhQKDHJlYWR5X2ltYWdlcxgIIAEoBBIbChNyZWFkeV9pbWFnZV9kaWdlc3RzGAkgAygJEhsKE3V0aWxfZGlza190b3RhbF9taWIYCiABKAQSGgoSdXRpbF9kaXNrX3VzZWRfbWliGAsgASgEEhoKEnV0aWxfbWVtX3RvdGFsX21pYhgMIAEoBBIZChF1dGlsX21lbV91c2VkX21pYhgNIAEoBBIUCgx1dGlsX2NwdV9wY3QYDiABKAISGQoRbGFzdF9oZWFydGJlYXRfYXQYDyABKAkSEAoIY29yZG9uZWQYECABKAgSFwoPYWxsb2NhdGFibGVfbWliGBEgASgEEhQKDHJlc2VydmVkX21pYhgSIAEoBBIQCghmcmVlX21pYhgTIAEoBBITCgt0b3RhbF92Y3B1cxgUIAEoDRIYChBjcHVfYnVkZ2V0X3ZjcHVzGBUgASgEEhYKDnJlc2VydmVkX3ZjcHVzGBYgASgEEhIKCmZyZWVfdmNwdXMYFyABKAQSGQoRdXRpbF9iYXNlX3NobV9taWIYGCABKAQSGwoTdXRpbF9wYXJrZWRfcHNzX21pYhgZIAEoBBIcChR1dGlsX3J1bm5pbmdfcHNzX21pYhgaIAEoBBIcChRmYWlsaW5nX2NhcGFiaWxpdGllcxgbIAMoCRIbChNmY19zbmFwc2hvdF92ZXJzaW9uGBwgASgJEhsKE2NhcGFiaWxpdGllc19zY2hlbWEYHSABKA0SGQoRbGl2ZV9tYXRlcmlhbGl6ZXMYHiABKA0SGQoRbGl2ZV9jYXB0dXJlX2pvYnMYHyABKA0SHwoXdXRpbF9jb21taXR0ZWRfc3dhcF9taWIYICABKAQSFAoMY29yZG9uX293bmVyGCEgASgJEjEKCnJldGlyZW1lbnQYIiABKAsyHS5lbmdyYW0uYXBwLnYxLkhvc3RSZXRpcmVtZW50SgQIBxAIUg9sb2NhbF9zbmFwc2hvdHMiIQoOR2V0SG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCSI4Cg9HZXRIb3N0UmVzcG9uc2USJQoEaG9zdBgBIAEoCzIXLmVuZ3JhbS5hcHAudjEuSG9zdFZpZXciKQoWR2V0SG9zdENvd1N0YXRlUmVxdWVzdBIPCgdob3N0X2lkGAEgASgJIlkKF0dldEhvc3RDb3dTdGF0ZVJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSLQoIc2Vzc2lvbnMYAiADKAsyGy5lbmdyYW0uYXBwLnYxLkNvd1N0YXRlVmlldyIjChBEcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkiEwoRRHJhaW5Ib3N0UmVzcG9uc2UiKAoVQWRtaW5EcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkiXgoWQWRtaW5EcmFpbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEg8KB3BsYW5uZWQYAiADKAkSEQoJZGVzY2VuZGVkGAMgAygJEg8KB3NraXBwZWQYBCABKA0iMwoRQ29yZG9uSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCRINCgVvd25lchgCIAEoCSI8ChdCZWdpbkhvc3RIYW5kb2ZmUmVxdWVzdBIPCgdob3N0X2lkGAEgASgJEhAKCHR0bF9zZWNzGAIgASgEIj0KGEJlZ2luSG9zdEhhbmRvZmZSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEhAKCGFjY2VwdGVkGAIgASgIIjUKEkNvcmRvbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEg4KBnN0YXR1cxgCIAEoCSI1ChNVbmNvcmRvbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkSDQoFb3duZXIYAiABKAkiNwoUVW5jb3Jkb25Ib3N0UmVzcG9uc2USDwoHaG9zdF9pZBgBIAEoCRIOCgZzdGF0dXMYAiABKAkiJAoRRGVsZXRlSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCSIUChJEZWxldGVIb3N0UmVzcG9uc2UiGgoYR2V0U3RvcmFnZVN1bW1hcnlSZXF1ZXN0IoQCChlHZXRTdG9yYWdlU3VtbWFyeVJlc3BvbnNlEhEKCXNuYXBzaG90cxgBIAEoBBIWCg5zbmFwc2hvdF9ieXRlcxgCIAEoBBISCgpnY19wZW5kaW5nGAMgASgEEhkKEXRyYWNrZWRfc2FuZGJveGVzGAQgASgEEhQKDGRpcnR5X2NodW5rcxgFIAEoBBIXCg91bmZsdXNoZWRfYnl0ZXMYBiABKAQSGAoQYXZnX2xvY2FsaXR5X3BjdBgHIAEoDRIqCgRyb3dzGAggAygLMhwuZW5ncmFtLmFwcC52MS5EdXJhYmlsaXR5Um93EhgKEGdjX3BlbmRpbmdfZXhhY3QYCSABKAgi5QEKDUR1cmFiaWxpdHlSb3cSEgoKc2FuZGJveF9pZBgBIAEoCRIXCgpzZXNzaW9uX2lkGAIgASgJSACIAQESDwoHaG9zdF9pZBgDIAEoCRIUCgxkaXJ0eV9jaHVua3MYBCABKA0SEwoLZGlydHlfYnl0ZXMYBSABKAQSEwoLYmFzZV9jaHVua3MYBiABKA0SGQoRYmFzZV9jaHVua3NfbG9jYWwYByABKA0SGgoNbGFzdF9mbHVzaF9hdBgIIAEoCUgBiAEBQg0KC19zZXNzaW9uX2lkQhAKDl9sYXN0X2ZsdXNoX2F0IikKE0ZsdXNoU2Vzc2lvblJlcXVlc3QSEgoKc2Vzc2lvbl9pZBgBIAEoCSJbChRGbHVzaFNlc3Npb25SZXNwb25zZRIPCgdvdXRjb21lGAEgASgJEh0KEG1hbmlmZXN0X3ZlcnNpb24YAiABKARIAIgBAUITChFfbWFuaWZlc3RfdmVyc2lvbiJWChZUZWxlcG9ydFNlc3Npb25SZXF1ZXN0EhIKCnNlc3Npb25faWQYASABKAkSGAoLdGFyZ2V0X2hvc3QYAiABKAlIAIgBAUIOCgxfdGFyZ2V0X2hvc3QiUgoXVGVsZXBvcnRTZXNzaW9uUmVzcG9uc2USEwoLdGVsZXBvcnRfaWQYASABKAkSDAoEa2luZBgCIAEoCRIUCgxkZXN0X2hvc3RfaWQYAyABKAkiFwoVR2V0RmxlZXREZW1hbmRSZXF1ZXN0IvEBChZHZXRGbGVldERlbWFuZFJlc3BvbnNlEhMKC3JlYWR5X2hvc3RzGAEgASgNEhkKEXNjaGVkdWxhYmxlX2hvc3RzGAIgASgNEhAKCGZyZWVfbWliGAMgASgEEhEKCXRvdGFsX21pYhgEIAEoBBISCgpmcmVlX3ZjcHVzGAUgASgEEhMKC3RvdGFsX3ZjcHVzGAYgASgEEhYKDmNvcmRvbmVkX2hvc3RzGAcgASgNEhcKD3F1ZXVlZF9zZXNzaW9ucxgIIAEoBBISCgpxdWV1ZWRfbWliGAkgASgEEhQKDHF1ZXVlZF92Y3B1cxgKIAEoBCJJCg5DaHVua0djUmVxdWVzdBIPCgdkcnlfcnVuGAEgASgIEhcKCmdyYWNlX3NlY3MYAiABKARIAIgBAUINCgtfZ3JhY2Vfc2VjcyKMAgoPQ2h1bmtHY1Jlc3BvbnNlEhUKDWxpc3RlZF9jaHVua3MYASABKAQSFgoObWFsZm9ybWVkX2tleXMYAiABKAQSFAoMcGluX3NldF9zaXplGAMgASgEEhkKEWNhbmRpZGF0ZXNfbWFya2VkGAQgASgEEhgKEHByb21vdGVkX2RlbGV0ZXMYByABKAQSHQoVcHJvbW90ZV9kZWxldGVfZXJyb3JzGAggASgEEhIKCmdyYWNlX3NlY3MYCSABKAQSGAoQZ2VuZXJhdGlvbl9tb3ZlZBgKIAEoCBIXCgptYXJrX2Vycm9yGAsgASgJSACIAQFCDQoLX21hcmtfZXJyb3JKBAgFEAZKBAgGEAciSgoPQnVuZGxlR2NSZXF1ZXN0Eg8KB2RyeV9ydW4YASABKAgSFwoKZ3JhY2Vfc2VjcxgCIAEoBEgAiAEBQg0KC19ncmFjZV9zZWNzIsMBChBCdW5kbGVHY1Jlc3BvbnNlEg4KBmxpc3RlZBgBIAEoBBIUCgxwaW5fc2V0X3NpemUYAiABKAQSGQoRY2FuZGlkYXRlc19tYXJrZWQYAyABKAQSGAoQcHJvbW90ZWRfZGVsZXRlcxgEIAEoBBIdChVwcm9tb3RlX2RlbGV0ZV9lcnJvcnMYBSABKAQSFQoNcmVzdGFydF9jb3VudBgGIAEoDRIeChZwcm9tb3RlX3JlcGlubmVkX3NraXBzGAcgASgEIlAKFVNuYXBzaG90QmxvYkdjUmVxdWVzdBIPCgdkcnlfcnVuGAEgASgIEhcKCmdyYWNlX3NlY3MYAiABKARIAIgBAUINCgtfZ3JhY2Vfc2VjcyLcAQoWU25hcHNob3RCbG9iR2NSZXNwb25zZRIOCgZsaXN0ZWQYASABKAQSEQoJbWFsZm9ybWVkGAIgASgEEhQKDHBpbl9zZXRfc2l6ZRgDIAEoBBIZChFjYW5kaWRhdGVzX21hcmtlZBgEIAEoBBIYChBwcm9tb3RlZF9kZWxldGVzGAUgASgEEh4KFnByb21vdGVfcmVwaW5uZWRfc2tpcHMYBiABKAQSHQoVcHJvbW90ZV9kZWxldGVfZXJyb3JzGAcgASgEEhUKDXJlc3RhcnRfY291bnQYCCABKA0iQwoRUmV0aXJlSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCRINCgVvd25lchgCIAEoCRIOCgZyZWFzb24YAyABKAkicwoSUmV0aXJlSG9zdFJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSMQoKcmV0aXJlbWVudBgCIAEoCzIdLmVuZ3JhbS5hcHAudjEuSG9zdFJldGlyZW1lbnQSGQoRdGVsZXBvcnRzX3BsYW5uZWQYAyABKA0ibgoOSG9zdFJldGlyZW1lbnQSFAoMcmVxdWVzdGVkX2F0GAEgASgJEhIKCnJldGlyZWRfYXQYAiABKAkSMgoIYmxvY2tlcnMYAyADKAsyIC5lbmdyYW0uYXBwLnYxLlJldGlyZW1lbnRCbG9ja2VyIjAKEVJldGlyZW1lbnRCbG9ja2VyEgwKBGtpbmQYASABKAkSDQoFY291bnQYAiABKAQy6AsKDEZsZWV0U2VydmljZRJOCglMaXN0SG9zdHMSHy5lbmdyYW0uYXBwLnYxLkxpc3RIb3N0c1JlcXVlc3QaIC5lbmdyYW0uYXBwLnYxLkxpc3RIb3N0c1Jlc3BvbnNlEkgKB0dldEhvc3QSHS5lbmdyYW0uYXBwLnYxLkdldEhvc3RSZXF1ZXN0Gh4uZW5ncmFtLmFwcC52MS5HZXRIb3N0UmVzcG9uc2USUQoKUmV0aXJlSG9zdBIgLmVuZ3JhbS5hcHAudjEuUmV0aXJlSG9zdFJlcXVlc3QaIS5lbmdyYW0uYXBwLnYxLlJldGlyZUhvc3RSZXNwb25zZRJgCg9HZXRIb3N0Q293U3RhdGUSJS5lbmdyYW0uYXBwLnYxLkdldEhvc3RDb3dTdGF0ZVJlcXVlc3QaJi5lbmdyYW0uYXBwLnYxLkdldEhvc3RDb3dTdGF0ZVJlc3BvbnNlEk4KCURyYWluSG9zdBIfLmVuZ3JhbS5hcHAudjEuRHJhaW5Ib3N0UmVxdWVzdBogLmVuZ3JhbS5hcHAudjEuRHJhaW5Ib3N0UmVzcG9uc2USXQoOQWRtaW5EcmFpbkhvc3QSJC5lbmdyYW0uYXBwLnYxLkFkbWluRHJhaW5Ib3N0UmVxdWVzdBolLmVuZ3JhbS5hcHAudjEuQWRtaW5EcmFpbkhvc3RSZXNwb25zZRJRCgpDb3Jkb25Ib3N0EiAuZW5ncmFtLmFwcC52MS5Db3Jkb25Ib3N0UmVxdWVzdBohLmVuZ3JhbS5hcHAudjEuQ29yZG9uSG9zdFJlc3BvbnNlEmMKEEJlZ2luSG9zdEhhbmRvZmYSJi5lbmdyYW0uYXBwLnYxLkJlZ2luSG9zdEhhbmRvZmZSZXF1ZXN0GicuZW5ncmFtLmFwcC52MS5CZWdpbkhvc3RIYW5kb2ZmUmVzcG9uc2USVwoMVW5jb3Jkb25Ib3N0EiIuZW5ncmFtLmFwcC52MS5VbmNvcmRvbkhvc3RSZXF1ZXN0GiMuZW5ncmFtLmFwcC52MS5VbmNvcmRvbkhvc3RSZXNwb25zZRJRCgpEZWxldGVIb3N0EiAuZW5ncmFtLmFwcC52MS5EZWxldGVIb3N0UmVxdWVzdBohLmVuZ3JhbS5hcHAudjEuRGVsZXRlSG9zdFJlc3BvbnNlEmYKEUdldFN0b3JhZ2VTdW1tYXJ5EicuZW5ncmFtLmFwcC52MS5HZXRTdG9yYWdlU3VtbWFyeVJlcXVlc3QaKC5lbmdyYW0uYXBwLnYxLkdldFN0b3JhZ2VTdW1tYXJ5UmVzcG9uc2USVwoMRmx1c2hTZXNzaW9uEiIuZW5ncmFtLmFwcC52MS5GbHVzaFNlc3Npb25SZXF1ZXN0GiMuZW5ncmFtLmFwcC52MS5GbHVzaFNlc3Npb25SZXNwb25zZRJgCg9UZWxlcG9ydFNlc3Npb24SJS5lbmdyYW0uYXBwLnYxLlRlbGVwb3J0U2Vzc2lvblJlcXVlc3QaJi5lbmdyYW0uYXBwLnYxLlRlbGVwb3J0U2Vzc2lvblJlc3BvbnNlEkgKB0NodW5rR2MSHS5lbmdyYW0uYXBwLnYxLkNodW5rR2NSZXF1ZXN0Gh4uZW5ncmFtLmFwcC52MS5DaHVua0djUmVzcG9uc2USSwoIQnVuZGxlR2MSHi5lbmdyYW0uYXBwLnYxLkJ1bmRsZUdjUmVxdWVzdBofLmVuZ3JhbS5hcHAudjEuQnVuZGxlR2NSZXNwb25zZRJdCg5TbmFwc2hvdEJsb2JHYxIkLmVuZ3JhbS5hcHAudjEuU25hcHNob3RCbG9iR2NSZXF1ZXN0GiUuZW5ncmFtLmFwcC52MS5TbmFwc2hvdEJsb2JHY1Jlc3BvbnNlEl0KDkdldEZsZWV0RGVtYW5kEiQuZW5ncmFtLmFwcC52MS5HZXRGbGVldERlbWFuZFJlcXVlc3QaJS5lbmdyYW0uYXBwLnYxLkdldEZsZWV0RGVtYW5kUmVzcG9uc2ViBnByb3RvMw", [file_engram_app_v1_session]); /** * @generated from message engram.app.v1.ListHostsRequest @@ -421,10 +421,7 @@ export const AdminDrainHostRequestSchema: GenMessage = /* messageDesc(file_engram_app_v1_fleet, 9); /** - * Mirrors api/admin.rs DrainHostResponse (renamed: this RPC carries the - * admin cordon+evacuate semantics; the soft status flip above keeps the - * plain DrainHost name). HTTP returns 202 — the evac_resumer scanner is - * the actual deliverable. + * Admission plans return immediately; the teleport machine drives each move. * * @generated from message engram.app.v1.AdminDrainHostResponse */ @@ -435,17 +432,19 @@ export type AdminDrainHostResponse = Message<"engram.app.v1.AdminDrainHostRespon hostId: string; /** - * Session ids now marked Evacuating; the scanner resumes each on a - * peer host within its sweep interval. - * - * @generated from field: repeated string evacuating = 2; + * @generated from field: repeated string planned = 2; + */ + planned: string[]; + + /** + * @generated from field: repeated string descended = 3; */ - evacuating: string[]; + descended: string[]; /** - * @generated from field: repeated engram.app.v1.DrainFailure failures = 3; + * @generated from field: uint32 skipped = 4; */ - failures: DrainFailure[]; + skipped: number; }; /** @@ -455,30 +454,6 @@ export type AdminDrainHostResponse = Message<"engram.app.v1.AdminDrainHostRespon export const AdminDrainHostResponseSchema: GenMessage = /*@__PURE__*/ messageDesc(file_engram_app_v1_fleet, 10); -/** - * @generated from message engram.app.v1.DrainFailure - */ -export type DrainFailure = Message<"engram.app.v1.DrainFailure"> & { - /** - * @generated from field: string session_id = 1; - */ - sessionId: string; - - /** - * Human-readable diagnostic; not machine-parsed, not stable. - * - * @generated from field: string error = 2; - */ - error: string; -}; - -/** - * Describes the message engram.app.v1.DrainFailure. - * Use `create(DrainFailureSchema)` to create a new message. - */ -export const DrainFailureSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 11); - /** * @generated from message engram.app.v1.CordonHostRequest */ @@ -502,7 +477,7 @@ export type CordonHostRequest = Message<"engram.app.v1.CordonHostRequest"> & { * Use `create(CordonHostRequestSchema)` to create a new message. */ export const CordonHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 12); + messageDesc(file_engram_app_v1_fleet, 11); /** * ADR 0116 A-D2: declare a planned handoff — extend the host's @@ -532,7 +507,7 @@ export type BeginHostHandoffRequest = Message<"engram.app.v1.BeginHostHandoffReq * Use `create(BeginHostHandoffRequestSchema)` to create a new message. */ export const BeginHostHandoffRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 13); + messageDesc(file_engram_app_v1_fleet, 12); /** * @generated from message engram.app.v1.BeginHostHandoffResponse @@ -558,7 +533,7 @@ export type BeginHostHandoffResponse = Message<"engram.app.v1.BeginHostHandoffRe * Use `create(BeginHostHandoffResponseSchema)` to create a new message. */ export const BeginHostHandoffResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 14); + messageDesc(file_engram_app_v1_fleet, 13); /** * @generated from message engram.app.v1.CordonHostResponse @@ -582,7 +557,7 @@ export type CordonHostResponse = Message<"engram.app.v1.CordonHostResponse"> & { * Use `create(CordonHostResponseSchema)` to create a new message. */ export const CordonHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 15); + messageDesc(file_engram_app_v1_fleet, 14); /** * @generated from message engram.app.v1.UncordonHostRequest @@ -606,7 +581,7 @@ export type UncordonHostRequest = Message<"engram.app.v1.UncordonHostRequest"> & * Use `create(UncordonHostRequestSchema)` to create a new message. */ export const UncordonHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 16); + messageDesc(file_engram_app_v1_fleet, 15); /** * Same Rust type as CordonHostResponse (CordonResponse); a distinct @@ -633,7 +608,7 @@ export type UncordonHostResponse = Message<"engram.app.v1.UncordonHostResponse"> * Use `create(UncordonHostResponseSchema)` to create a new message. */ export const UncordonHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 17); + messageDesc(file_engram_app_v1_fleet, 16); /** * @generated from message engram.app.v1.DeleteHostRequest @@ -650,7 +625,7 @@ export type DeleteHostRequest = Message<"engram.app.v1.DeleteHostRequest"> & { * Use `create(DeleteHostRequestSchema)` to create a new message. */ export const DeleteHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 18); + messageDesc(file_engram_app_v1_fleet, 17); /** * Mirrors the old `DELETE /api/admin/hosts/:id` (204 No Content) — the @@ -667,7 +642,7 @@ export type DeleteHostResponse = Message<"engram.app.v1.DeleteHostResponse"> & { * Use `create(DeleteHostResponseSchema)` to create a new message. */ export const DeleteHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 19); + messageDesc(file_engram_app_v1_fleet, 18); /** * @generated from message engram.app.v1.GetStorageSummaryRequest @@ -680,7 +655,7 @@ export type GetStorageSummaryRequest = Message<"engram.app.v1.GetStorageSummaryR * Use `create(GetStorageSummaryRequestSchema)` to create a new message. */ export const GetStorageSummaryRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 20); + messageDesc(file_engram_app_v1_fleet, 19); /** * Mirrors api/storage.rs StorageSummaryResponse (ADR 0029): fleet-wide @@ -760,7 +735,7 @@ export type GetStorageSummaryResponse = Message<"engram.app.v1.GetStorageSummary * Use `create(GetStorageSummaryResponseSchema)` to create a new message. */ export const GetStorageSummaryResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 21); + messageDesc(file_engram_app_v1_fleet, 20); /** * One per-sandbox row of the durability ledger. @@ -824,7 +799,7 @@ export type DurabilityRow = Message<"engram.app.v1.DurabilityRow"> & { * Use `create(DurabilityRowSchema)` to create a new message. */ export const DurabilityRowSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 22); + messageDesc(file_engram_app_v1_fleet, 21); /** * @generated from message engram.app.v1.FlushSessionRequest @@ -841,7 +816,7 @@ export type FlushSessionRequest = Message<"engram.app.v1.FlushSessionRequest"> & * Use `create(FlushSessionRequestSchema)` to create a new message. */ export const FlushSessionRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 23); + messageDesc(file_engram_app_v1_fleet, 22); /** * Mirrors api/admin.rs FlushNowResult. @@ -873,64 +848,56 @@ export type FlushSessionResponse = Message<"engram.app.v1.FlushSessionResponse"> * Use `create(FlushSessionResponseSchema)` to create a new message. */ export const FlushSessionResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 24); + messageDesc(file_engram_app_v1_fleet, 23); /** - * @generated from message engram.app.v1.EvacuateSessionRequest + * @generated from message engram.app.v1.TeleportSessionRequest */ -export type EvacuateSessionRequest = Message<"engram.app.v1.EvacuateSessionRequest"> & { +export type TeleportSessionRequest = Message<"engram.app.v1.TeleportSessionRequest"> & { /** * @generated from field: string session_id = 1; */ sessionId: string; /** - * Reserved for a future operator override; carried on today's HTTP - * body (api/admin.rs EvacuateSessionRequest.target_host) but IGNORED - * by the handler — the scanner picks any non-source host via the - * standard policy. - * * @generated from field: optional string target_host = 2; */ targetHost?: string; }; /** - * Describes the message engram.app.v1.EvacuateSessionRequest. - * Use `create(EvacuateSessionRequestSchema)` to create a new message. + * Describes the message engram.app.v1.TeleportSessionRequest. + * Use `create(TeleportSessionRequestSchema)` to create a new message. */ -export const EvacuateSessionRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 25); +export const TeleportSessionRequestSchema: GenMessage = /*@__PURE__*/ + messageDesc(file_engram_app_v1_fleet, 24); /** - * Mirrors api/admin.rs EvacuateSessionResponse. HTTP returns 202; the - * evac_resumer scanner completes the transition. - * - * @generated from message engram.app.v1.EvacuateSessionResponse + * @generated from message engram.app.v1.TeleportSessionResponse */ -export type EvacuateSessionResponse = Message<"engram.app.v1.EvacuateSessionResponse"> & { +export type TeleportSessionResponse = Message<"engram.app.v1.TeleportSessionResponse"> & { /** - * @generated from field: string session_id = 1; + * @generated from field: string teleport_id = 1; */ - sessionId: string; + teleportId: string; /** - * Always "evacuating" on success — the session is paused, - * snapshotted, and the scanner will resume it on a peer within the - * next sweep interval (≤10s default). Watch the session's event - * stream for the Evacuating → Created → Active chain. - * - * @generated from field: string status = 2; + * @generated from field: string kind = 2; */ - status: string; + kind: string; + + /** + * @generated from field: string dest_host_id = 3; + */ + destHostId: string; }; /** - * Describes the message engram.app.v1.EvacuateSessionResponse. - * Use `create(EvacuateSessionResponseSchema)` to create a new message. + * Describes the message engram.app.v1.TeleportSessionResponse. + * Use `create(TeleportSessionResponseSchema)` to create a new message. */ -export const EvacuateSessionResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 26); +export const TeleportSessionResponseSchema: GenMessage = /*@__PURE__*/ + messageDesc(file_engram_app_v1_fleet, 25); /** * @generated from message engram.app.v1.GetFleetDemandRequest @@ -943,7 +910,7 @@ export type GetFleetDemandRequest = Message<"engram.app.v1.GetFleetDemandRequest * Use `create(GetFleetDemandRequestSchema)` to create a new message. */ export const GetFleetDemandRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 27); + messageDesc(file_engram_app_v1_fleet, 26); /** * Mirrors api/admin.rs FleetDemandResponse (ADR 0044 K4 / 0047 / 0048). The @@ -1028,7 +995,7 @@ export type GetFleetDemandResponse = Message<"engram.app.v1.GetFleetDemandRespon * Use `create(GetFleetDemandResponseSchema)` to create a new message. */ export const GetFleetDemandResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 28); + messageDesc(file_engram_app_v1_fleet, 27); /** * @generated from message engram.app.v1.ChunkGcRequest @@ -1065,7 +1032,7 @@ export type ChunkGcRequest = Message<"engram.app.v1.ChunkGcRequest"> & { * Use `create(ChunkGcRequestSchema)` to create a new message. */ export const ChunkGcRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 29); + messageDesc(file_engram_app_v1_fleet, 28); /** * Mirrors api/admin.rs ChunkGcSweepResult. @@ -1144,7 +1111,7 @@ export type ChunkGcResponse = Message<"engram.app.v1.ChunkGcResponse"> & { * Use `create(ChunkGcResponseSchema)` to create a new message. */ export const ChunkGcResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 30); + messageDesc(file_engram_app_v1_fleet, 29); /** * @generated from message engram.app.v1.BundleGcRequest @@ -1169,7 +1136,7 @@ export type BundleGcRequest = Message<"engram.app.v1.BundleGcRequest"> & { * Use `create(BundleGcRequestSchema)` to create a new message. */ export const BundleGcRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 31); + messageDesc(file_engram_app_v1_fleet, 30); /** * Mirrors bundle_gc.rs BundleSweepReport (ADR 0035 §5). Unlike the @@ -1226,7 +1193,7 @@ export type BundleGcResponse = Message<"engram.app.v1.BundleGcResponse"> & { * Use `create(BundleGcResponseSchema)` to create a new message. */ export const BundleGcResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 32); + messageDesc(file_engram_app_v1_fleet, 31); /** * @generated from message engram.app.v1.SnapshotBlobGcRequest @@ -1251,7 +1218,7 @@ export type SnapshotBlobGcRequest = Message<"engram.app.v1.SnapshotBlobGcRequest * Use `create(SnapshotBlobGcRequestSchema)` to create a new message. */ export const SnapshotBlobGcRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 33); + messageDesc(file_engram_app_v1_fleet, 32); /** * Mirrors snapshot_blob_gc.rs SnapshotBlobSweepReport (ADR 0028 @@ -1314,7 +1281,7 @@ export type SnapshotBlobGcResponse = Message<"engram.app.v1.SnapshotBlobGcRespon * Use `create(SnapshotBlobGcResponseSchema)` to create a new message. */ export const SnapshotBlobGcResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 34); + messageDesc(file_engram_app_v1_fleet, 33); /** * ADR 0123 A3: the idempotent retirement request. Cordons the host with @@ -1350,7 +1317,7 @@ export type RetireHostRequest = Message<"engram.app.v1.RetireHostRequest"> & { * Use `create(RetireHostRequestSchema)` to create a new message. */ export const RetireHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 35); + messageDesc(file_engram_app_v1_fleet, 34); /** * @generated from message engram.app.v1.RetireHostResponse @@ -1379,7 +1346,7 @@ export type RetireHostResponse = Message<"engram.app.v1.RetireHostResponse"> & { * Use `create(RetireHostResponseSchema)` to create a new message. */ export const RetireHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 36); + messageDesc(file_engram_app_v1_fleet, 35); /** * @generated from message engram.app.v1.HostRetirement @@ -1413,7 +1380,7 @@ export type HostRetirement = Message<"engram.app.v1.HostRetirement"> & { * Use `create(HostRetirementSchema)` to create a new message. */ export const HostRetirementSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 37); + messageDesc(file_engram_app_v1_fleet, 36); /** * @generated from message engram.app.v1.RetirementBlocker @@ -1435,7 +1402,7 @@ export type RetirementBlocker = Message<"engram.app.v1.RetirementBlocker"> & { * Use `create(RetirementBlockerSchema)` to create a new message. */ export const RetirementBlockerSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 38); + messageDesc(file_engram_app_v1_fleet, 37); /** * Fleet, storage, and GC operations (ADR 0051 §2.3, rev 2026-06-10). @@ -1572,14 +1539,15 @@ export const FleetService: GenService<{ output: typeof FlushSessionResponseSchema; }, /** - * ADR 0018 async evacuation (POST /api/admin/sessions/:id/evacuate). + * ADR 0123 B: admit one teleport for a session (synchronous admission; + * the durable machine drives it). Replaces the ADR 0018 evacuate route. * - * @generated from rpc engram.app.v1.FleetService.EvacuateSession + * @generated from rpc engram.app.v1.FleetService.TeleportSession */ - evacuateSession: { + teleportSession: { methodKind: "unary"; - input: typeof EvacuateSessionRequestSchema; - output: typeof EvacuateSessionResponseSchema; + input: typeof TeleportSessionRequestSchema; + output: typeof TeleportSessionResponseSchema; }, /** * The three GC sweeps (ADR 0016 Phase C / ADR 0035 §5 / ADR 0028 diff --git a/orchestrator/src/listeners/__tests__/frame-taxonomy.test.ts b/orchestrator/src/listeners/__tests__/frame-taxonomy.test.ts index 348e6f47e..7ec912d9d 100644 --- a/orchestrator/src/listeners/__tests__/frame-taxonomy.test.ts +++ b/orchestrator/src/listeners/__tests__/frame-taxonomy.test.ts @@ -62,6 +62,8 @@ const WIRE_FRAME_KINDS: Readonly> = { run_completed: "durable", run_interrupted: "durable", run_continued: "durable", + // ADR 0123: the result of a session teleport. + teleport_finished: "durable", harness_idle: "durable", prompt_queued: "durable", prompt_edited: "durable", diff --git a/orchestrator/src/routes/admin.ts b/orchestrator/src/routes/admin.ts index 340fce265..33cea20d6 100644 --- a/orchestrator/src/routes/admin.ts +++ b/orchestrator/src/routes/admin.ts @@ -20,7 +20,7 @@ * are promoted to gRPC in a future task, delete these routes and migrate the * hooks to connect-query. * - * Note: teleportSession uses FleetService.EvacuateSession (gRPC, already + * Note: teleportSession uses FleetService.TeleportSession (gRPC, already * passing through the CASL gate) — no proxy route needed for it. * * Production auth note (ADR 0051 Task 32): The forwarded bearer is diff --git a/web/src/components/SessionDiagnostics.tsx b/web/src/components/SessionDiagnostics.tsx index 4da3847ab..ffb04f9ff 100644 --- a/web/src/components/SessionDiagnostics.tsx +++ b/web/src/components/SessionDiagnostics.tsx @@ -237,7 +237,7 @@ export function DiagnosticsPanel({ {/* Admin live-ops — self-gating (admin + Active only), so a non-admin sees a clean read-only ledger above and nothing here. */} - {session && } + {session && } {session && } @@ -287,24 +287,39 @@ function summarizeRaw(ev: unknown): string { // Admin live-ops (ADR 0045 Phase F) — relocated from the session-detail rail. // --------------------------------------------------------------------------- -// Teleport (live-migrate) an Active session onto a chosen host. Today the verb -// rides the snapshot-rehome evac pipeline (a brief pause); ADR 0045 Phase C -// swaps it to post-copy live migration under the same control. Renders nothing -// unless the viewer is an admin and the session is Active (the only relocatable -// state). -function TeleportControl({ session }: { session: Session }) { +// Admission returns the move identity; the durable event marks its outcome. +function TeleportControl({ session, events }: { session: Session; events: IndexedEvent[] }) { const isAdmin = useIsAdmin(); const { data: hosts } = useHosts(); const teleport = useTeleportSession(session.id); const [target, setTarget] = useState(""); - if (!isAdmin || session.status !== "active") return null; + const finished = events + .map((item) => item.event) + .reverse() + .find( + (event) => + event.type === "teleport_finished" && event.teleport_id === teleport.data?.teleportId, + ); + if (!isAdmin || (session.status !== "active" && !teleport.data)) return null; const candidates = (hosts ?? []).filter((h) => h.status === "ready" && h.id !== session.host_id); return (
Teleport + {teleport.error && ( + + {teleport.error.message} + + )} + {teleport.data && ( + + {teleport.data.kind} to {teleport.data.destHostId}:{" "} + {finished?.type === "teleport_finished" ? finished.outcome : "in progress"} + {finished?.type === "teleport_finished" && finished.error ? ` — ${finished.error}` : ""} + + )} {candidates.length === 0 ? ( No other ready host is available. @@ -331,7 +346,12 @@ function TeleportControl({ session }: { session: Session }) { size="sm" variant="secondary" className="w-full" - disabled={!target || teleport.isPending} + disabled={ + !target || + teleport.isPending || + session.status !== "active" || + (!!teleport.data && !finished) + } onClick={() => teleport.mutate(target)} > {teleport.isPending ? "Teleporting…" : "Teleport"} diff --git a/web/src/events.ts b/web/src/events.ts index 86f4b9e6c..85bef7595 100644 --- a/web/src/events.ts +++ b/web/src/events.ts @@ -60,6 +60,15 @@ export type FileChange = // `serde(tag = "type", rename_all = "snake_case")` produces a discriminated // union with `type` as the discriminant. export type SessionEvent = + | { + type: "teleport_finished"; + teleport_id: string; + outcome: "done" | "aborted" | "failed"; + kind: "live" | "snapshot"; + dest_host_id: string | null; + error: string | null; + at: string; + } | { type: "status_changed"; from: SessionState; diff --git a/web/src/gen/engram/app/v1/fleet-FleetService_connectquery.ts b/web/src/gen/engram/app/v1/fleet-FleetService_connectquery.ts index 520e271ec..21a17ff17 100644 --- a/web/src/gen/engram/app/v1/fleet-FleetService_connectquery.ts +++ b/web/src/gen/engram/app/v1/fleet-FleetService_connectquery.ts @@ -91,11 +91,12 @@ export const getStorageSummary = FleetService.method.getStorageSummary; export const flushSession = FleetService.method.flushSession; /** - * ADR 0018 async evacuation (POST /api/admin/sessions/:id/evacuate). + * ADR 0123 B: admit one teleport for a session (synchronous admission; + * the durable machine drives it). Replaces the ADR 0018 evacuate route. * - * @generated from rpc engram.app.v1.FleetService.EvacuateSession + * @generated from rpc engram.app.v1.FleetService.TeleportSession */ -export const evacuateSession = FleetService.method.evacuateSession; +export const teleportSession = FleetService.method.teleportSession; /** * The three GC sweeps (ADR 0016 Phase C / ADR 0035 §5 / ADR 0028 diff --git a/web/src/gen/engram/app/v1/fleet_pb.ts b/web/src/gen/engram/app/v1/fleet_pb.ts index 7c7e01ea0..7da919176 100644 --- a/web/src/gen/engram/app/v1/fleet_pb.ts +++ b/web/src/gen/engram/app/v1/fleet_pb.ts @@ -12,7 +12,7 @@ import type { Message } from "@bufbuild/protobuf"; * Describes the file engram/app/v1/fleet.proto. */ export const file_engram_app_v1_fleet: GenFile = /*@__PURE__*/ - fileDesc("ChllbmdyYW0vYXBwL3YxL2ZsZWV0LnByb3RvEg1lbmdyYW0uYXBwLnYxIhIKEExpc3RIb3N0c1JlcXVlc3QiOwoRTGlzdEhvc3RzUmVzcG9uc2USJgoFaG9zdHMYASADKAsyFy5lbmdyYW0uYXBwLnYxLkhvc3RWaWV3IvEGCghIb3N0VmlldxIKCgJpZBgBIAEoCRIQCghob3N0bmFtZRgCIAEoCRIOCgZzdGF0dXMYAyABKAkSGgoSY2FwYWNpdHlfdG90YWxfbWliGAQgASgEEhkKEWNhcGFjaXR5X3VzZWRfbWliGAUgASgEEhkKEXJ1bm5pbmdfc2FuZGJveGVzGAYgASgNEhQKDHJlYWR5X2ltYWdlcxgIIAEoBBIbChNyZWFkeV9pbWFnZV9kaWdlc3RzGAkgAygJEhsKE3V0aWxfZGlza190b3RhbF9taWIYCiABKAQSGgoSdXRpbF9kaXNrX3VzZWRfbWliGAsgASgEEhoKEnV0aWxfbWVtX3RvdGFsX21pYhgMIAEoBBIZChF1dGlsX21lbV91c2VkX21pYhgNIAEoBBIUCgx1dGlsX2NwdV9wY3QYDiABKAISGQoRbGFzdF9oZWFydGJlYXRfYXQYDyABKAkSEAoIY29yZG9uZWQYECABKAgSFwoPYWxsb2NhdGFibGVfbWliGBEgASgEEhQKDHJlc2VydmVkX21pYhgSIAEoBBIQCghmcmVlX21pYhgTIAEoBBITCgt0b3RhbF92Y3B1cxgUIAEoDRIYChBjcHVfYnVkZ2V0X3ZjcHVzGBUgASgEEhYKDnJlc2VydmVkX3ZjcHVzGBYgASgEEhIKCmZyZWVfdmNwdXMYFyABKAQSGQoRdXRpbF9iYXNlX3NobV9taWIYGCABKAQSGwoTdXRpbF9wYXJrZWRfcHNzX21pYhgZIAEoBBIcChR1dGlsX3J1bm5pbmdfcHNzX21pYhgaIAEoBBIcChRmYWlsaW5nX2NhcGFiaWxpdGllcxgbIAMoCRIbChNmY19zbmFwc2hvdF92ZXJzaW9uGBwgASgJEhsKE2NhcGFiaWxpdGllc19zY2hlbWEYHSABKA0SGQoRbGl2ZV9tYXRlcmlhbGl6ZXMYHiABKA0SGQoRbGl2ZV9jYXB0dXJlX2pvYnMYHyABKA0SHwoXdXRpbF9jb21taXR0ZWRfc3dhcF9taWIYICABKAQSFAoMY29yZG9uX293bmVyGCEgASgJEjEKCnJldGlyZW1lbnQYIiABKAsyHS5lbmdyYW0uYXBwLnYxLkhvc3RSZXRpcmVtZW50SgQIBxAIUg9sb2NhbF9zbmFwc2hvdHMiIQoOR2V0SG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCSI4Cg9HZXRIb3N0UmVzcG9uc2USJQoEaG9zdBgBIAEoCzIXLmVuZ3JhbS5hcHAudjEuSG9zdFZpZXciKQoWR2V0SG9zdENvd1N0YXRlUmVxdWVzdBIPCgdob3N0X2lkGAEgASgJIlkKF0dldEhvc3RDb3dTdGF0ZVJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSLQoIc2Vzc2lvbnMYAiADKAsyGy5lbmdyYW0uYXBwLnYxLkNvd1N0YXRlVmlldyIjChBEcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkiEwoRRHJhaW5Ib3N0UmVzcG9uc2UiKAoVQWRtaW5EcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkibAoWQWRtaW5EcmFpbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEhIKCmV2YWN1YXRpbmcYAiADKAkSLQoIZmFpbHVyZXMYAyADKAsyGy5lbmdyYW0uYXBwLnYxLkRyYWluRmFpbHVyZSIxCgxEcmFpbkZhaWx1cmUSEgoKc2Vzc2lvbl9pZBgBIAEoCRINCgVlcnJvchgCIAEoCSIzChFDb3Jkb25Ib3N0UmVxdWVzdBIPCgdob3N0X2lkGAEgASgJEg0KBW93bmVyGAIgASgJIjwKF0JlZ2luSG9zdEhhbmRvZmZSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkSEAoIdHRsX3NlY3MYAiABKAQiPQoYQmVnaW5Ib3N0SGFuZG9mZlJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSEAoIYWNjZXB0ZWQYAiABKAgiNQoSQ29yZG9uSG9zdFJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSDgoGc3RhdHVzGAIgASgJIjUKE1VuY29yZG9uSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCRINCgVvd25lchgCIAEoCSI3ChRVbmNvcmRvbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEg4KBnN0YXR1cxgCIAEoCSIkChFEZWxldGVIb3N0UmVxdWVzdBIPCgdob3N0X2lkGAEgASgJIhQKEkRlbGV0ZUhvc3RSZXNwb25zZSIaChhHZXRTdG9yYWdlU3VtbWFyeVJlcXVlc3QihAIKGUdldFN0b3JhZ2VTdW1tYXJ5UmVzcG9uc2USEQoJc25hcHNob3RzGAEgASgEEhYKDnNuYXBzaG90X2J5dGVzGAIgASgEEhIKCmdjX3BlbmRpbmcYAyABKAQSGQoRdHJhY2tlZF9zYW5kYm94ZXMYBCABKAQSFAoMZGlydHlfY2h1bmtzGAUgASgEEhcKD3VuZmx1c2hlZF9ieXRlcxgGIAEoBBIYChBhdmdfbG9jYWxpdHlfcGN0GAcgASgNEioKBHJvd3MYCCADKAsyHC5lbmdyYW0uYXBwLnYxLkR1cmFiaWxpdHlSb3cSGAoQZ2NfcGVuZGluZ19leGFjdBgJIAEoCCLlAQoNRHVyYWJpbGl0eVJvdxISCgpzYW5kYm94X2lkGAEgASgJEhcKCnNlc3Npb25faWQYAiABKAlIAIgBARIPCgdob3N0X2lkGAMgASgJEhQKDGRpcnR5X2NodW5rcxgEIAEoDRITCgtkaXJ0eV9ieXRlcxgFIAEoBBITCgtiYXNlX2NodW5rcxgGIAEoDRIZChFiYXNlX2NodW5rc19sb2NhbBgHIAEoDRIaCg1sYXN0X2ZsdXNoX2F0GAggASgJSAGIAQFCDQoLX3Nlc3Npb25faWRCEAoOX2xhc3RfZmx1c2hfYXQiKQoTRmx1c2hTZXNzaW9uUmVxdWVzdBISCgpzZXNzaW9uX2lkGAEgASgJIlsKFEZsdXNoU2Vzc2lvblJlc3BvbnNlEg8KB291dGNvbWUYASABKAkSHQoQbWFuaWZlc3RfdmVyc2lvbhgCIAEoBEgAiAEBQhMKEV9tYW5pZmVzdF92ZXJzaW9uIlYKFkV2YWN1YXRlU2Vzc2lvblJlcXVlc3QSEgoKc2Vzc2lvbl9pZBgBIAEoCRIYCgt0YXJnZXRfaG9zdBgCIAEoCUgAiAEBQg4KDF90YXJnZXRfaG9zdCI9ChdFdmFjdWF0ZVNlc3Npb25SZXNwb25zZRISCgpzZXNzaW9uX2lkGAEgASgJEg4KBnN0YXR1cxgCIAEoCSIXChVHZXRGbGVldERlbWFuZFJlcXVlc3Qi8QEKFkdldEZsZWV0RGVtYW5kUmVzcG9uc2USEwoLcmVhZHlfaG9zdHMYASABKA0SGQoRc2NoZWR1bGFibGVfaG9zdHMYAiABKA0SEAoIZnJlZV9taWIYAyABKAQSEQoJdG90YWxfbWliGAQgASgEEhIKCmZyZWVfdmNwdXMYBSABKAQSEwoLdG90YWxfdmNwdXMYBiABKAQSFgoOY29yZG9uZWRfaG9zdHMYByABKA0SFwoPcXVldWVkX3Nlc3Npb25zGAggASgEEhIKCnF1ZXVlZF9taWIYCSABKAQSFAoMcXVldWVkX3ZjcHVzGAogASgEIkkKDkNodW5rR2NSZXF1ZXN0Eg8KB2RyeV9ydW4YASABKAgSFwoKZ3JhY2Vfc2VjcxgCIAEoBEgAiAEBQg0KC19ncmFjZV9zZWNzIowCCg9DaHVua0djUmVzcG9uc2USFQoNbGlzdGVkX2NodW5rcxgBIAEoBBIWCg5tYWxmb3JtZWRfa2V5cxgCIAEoBBIUCgxwaW5fc2V0X3NpemUYAyABKAQSGQoRY2FuZGlkYXRlc19tYXJrZWQYBCABKAQSGAoQcHJvbW90ZWRfZGVsZXRlcxgHIAEoBBIdChVwcm9tb3RlX2RlbGV0ZV9lcnJvcnMYCCABKAQSEgoKZ3JhY2Vfc2VjcxgJIAEoBBIYChBnZW5lcmF0aW9uX21vdmVkGAogASgIEhcKCm1hcmtfZXJyb3IYCyABKAlIAIgBAUINCgtfbWFya19lcnJvckoECAUQBkoECAYQByJKCg9CdW5kbGVHY1JlcXVlc3QSDwoHZHJ5X3J1bhgBIAEoCBIXCgpncmFjZV9zZWNzGAIgASgESACIAQFCDQoLX2dyYWNlX3NlY3MiwwEKEEJ1bmRsZUdjUmVzcG9uc2USDgoGbGlzdGVkGAEgASgEEhQKDHBpbl9zZXRfc2l6ZRgCIAEoBBIZChFjYW5kaWRhdGVzX21hcmtlZBgDIAEoBBIYChBwcm9tb3RlZF9kZWxldGVzGAQgASgEEh0KFXByb21vdGVfZGVsZXRlX2Vycm9ycxgFIAEoBBIVCg1yZXN0YXJ0X2NvdW50GAYgASgNEh4KFnByb21vdGVfcmVwaW5uZWRfc2tpcHMYByABKAQiUAoVU25hcHNob3RCbG9iR2NSZXF1ZXN0Eg8KB2RyeV9ydW4YASABKAgSFwoKZ3JhY2Vfc2VjcxgCIAEoBEgAiAEBQg0KC19ncmFjZV9zZWNzItwBChZTbmFwc2hvdEJsb2JHY1Jlc3BvbnNlEg4KBmxpc3RlZBgBIAEoBBIRCgltYWxmb3JtZWQYAiABKAQSFAoMcGluX3NldF9zaXplGAMgASgEEhkKEWNhbmRpZGF0ZXNfbWFya2VkGAQgASgEEhgKEHByb21vdGVkX2RlbGV0ZXMYBSABKAQSHgoWcHJvbW90ZV9yZXBpbm5lZF9za2lwcxgGIAEoBBIdChVwcm9tb3RlX2RlbGV0ZV9lcnJvcnMYByABKAQSFQoNcmVzdGFydF9jb3VudBgIIAEoDSJDChFSZXRpcmVIb3N0UmVxdWVzdBIPCgdob3N0X2lkGAEgASgJEg0KBW93bmVyGAIgASgJEg4KBnJlYXNvbhgDIAEoCSJzChJSZXRpcmVIb3N0UmVzcG9uc2USDwoHaG9zdF9pZBgBIAEoCRIxCgpyZXRpcmVtZW50GAIgASgLMh0uZW5ncmFtLmFwcC52MS5Ib3N0UmV0aXJlbWVudBIZChF0ZWxlcG9ydHNfcGxhbm5lZBgDIAEoDSJuCg5Ib3N0UmV0aXJlbWVudBIUCgxyZXF1ZXN0ZWRfYXQYASABKAkSEgoKcmV0aXJlZF9hdBgCIAEoCRIyCghibG9ja2VycxgDIAMoCzIgLmVuZ3JhbS5hcHAudjEuUmV0aXJlbWVudEJsb2NrZXIiMAoRUmV0aXJlbWVudEJsb2NrZXISDAoEa2luZBgBIAEoCRINCgVjb3VudBgCIAEoBDLoCwoMRmxlZXRTZXJ2aWNlEk4KCUxpc3RIb3N0cxIfLmVuZ3JhbS5hcHAudjEuTGlzdEhvc3RzUmVxdWVzdBogLmVuZ3JhbS5hcHAudjEuTGlzdEhvc3RzUmVzcG9uc2USSAoHR2V0SG9zdBIdLmVuZ3JhbS5hcHAudjEuR2V0SG9zdFJlcXVlc3QaHi5lbmdyYW0uYXBwLnYxLkdldEhvc3RSZXNwb25zZRJRCgpSZXRpcmVIb3N0EiAuZW5ncmFtLmFwcC52MS5SZXRpcmVIb3N0UmVxdWVzdBohLmVuZ3JhbS5hcHAudjEuUmV0aXJlSG9zdFJlc3BvbnNlEmAKD0dldEhvc3RDb3dTdGF0ZRIlLmVuZ3JhbS5hcHAudjEuR2V0SG9zdENvd1N0YXRlUmVxdWVzdBomLmVuZ3JhbS5hcHAudjEuR2V0SG9zdENvd1N0YXRlUmVzcG9uc2USTgoJRHJhaW5Ib3N0Eh8uZW5ncmFtLmFwcC52MS5EcmFpbkhvc3RSZXF1ZXN0GiAuZW5ncmFtLmFwcC52MS5EcmFpbkhvc3RSZXNwb25zZRJdCg5BZG1pbkRyYWluSG9zdBIkLmVuZ3JhbS5hcHAudjEuQWRtaW5EcmFpbkhvc3RSZXF1ZXN0GiUuZW5ncmFtLmFwcC52MS5BZG1pbkRyYWluSG9zdFJlc3BvbnNlElEKCkNvcmRvbkhvc3QSIC5lbmdyYW0uYXBwLnYxLkNvcmRvbkhvc3RSZXF1ZXN0GiEuZW5ncmFtLmFwcC52MS5Db3Jkb25Ib3N0UmVzcG9uc2USYwoQQmVnaW5Ib3N0SGFuZG9mZhImLmVuZ3JhbS5hcHAudjEuQmVnaW5Ib3N0SGFuZG9mZlJlcXVlc3QaJy5lbmdyYW0uYXBwLnYxLkJlZ2luSG9zdEhhbmRvZmZSZXNwb25zZRJXCgxVbmNvcmRvbkhvc3QSIi5lbmdyYW0uYXBwLnYxLlVuY29yZG9uSG9zdFJlcXVlc3QaIy5lbmdyYW0uYXBwLnYxLlVuY29yZG9uSG9zdFJlc3BvbnNlElEKCkRlbGV0ZUhvc3QSIC5lbmdyYW0uYXBwLnYxLkRlbGV0ZUhvc3RSZXF1ZXN0GiEuZW5ncmFtLmFwcC52MS5EZWxldGVIb3N0UmVzcG9uc2USZgoRR2V0U3RvcmFnZVN1bW1hcnkSJy5lbmdyYW0uYXBwLnYxLkdldFN0b3JhZ2VTdW1tYXJ5UmVxdWVzdBooLmVuZ3JhbS5hcHAudjEuR2V0U3RvcmFnZVN1bW1hcnlSZXNwb25zZRJXCgxGbHVzaFNlc3Npb24SIi5lbmdyYW0uYXBwLnYxLkZsdXNoU2Vzc2lvblJlcXVlc3QaIy5lbmdyYW0uYXBwLnYxLkZsdXNoU2Vzc2lvblJlc3BvbnNlEmAKD0V2YWN1YXRlU2Vzc2lvbhIlLmVuZ3JhbS5hcHAudjEuRXZhY3VhdGVTZXNzaW9uUmVxdWVzdBomLmVuZ3JhbS5hcHAudjEuRXZhY3VhdGVTZXNzaW9uUmVzcG9uc2USSAoHQ2h1bmtHYxIdLmVuZ3JhbS5hcHAudjEuQ2h1bmtHY1JlcXVlc3QaHi5lbmdyYW0uYXBwLnYxLkNodW5rR2NSZXNwb25zZRJLCghCdW5kbGVHYxIeLmVuZ3JhbS5hcHAudjEuQnVuZGxlR2NSZXF1ZXN0Gh8uZW5ncmFtLmFwcC52MS5CdW5kbGVHY1Jlc3BvbnNlEl0KDlNuYXBzaG90QmxvYkdjEiQuZW5ncmFtLmFwcC52MS5TbmFwc2hvdEJsb2JHY1JlcXVlc3QaJS5lbmdyYW0uYXBwLnYxLlNuYXBzaG90QmxvYkdjUmVzcG9uc2USXQoOR2V0RmxlZXREZW1hbmQSJC5lbmdyYW0uYXBwLnYxLkdldEZsZWV0RGVtYW5kUmVxdWVzdBolLmVuZ3JhbS5hcHAudjEuR2V0RmxlZXREZW1hbmRSZXNwb25zZWIGcHJvdG8z", [file_engram_app_v1_session]); + fileDesc("ChllbmdyYW0vYXBwL3YxL2ZsZWV0LnByb3RvEg1lbmdyYW0uYXBwLnYxIhIKEExpc3RIb3N0c1JlcXVlc3QiOwoRTGlzdEhvc3RzUmVzcG9uc2USJgoFaG9zdHMYASADKAsyFy5lbmdyYW0uYXBwLnYxLkhvc3RWaWV3IvEGCghIb3N0VmlldxIKCgJpZBgBIAEoCRIQCghob3N0bmFtZRgCIAEoCRIOCgZzdGF0dXMYAyABKAkSGgoSY2FwYWNpdHlfdG90YWxfbWliGAQgASgEEhkKEWNhcGFjaXR5X3VzZWRfbWliGAUgASgEEhkKEXJ1bm5pbmdfc2FuZGJveGVzGAYgASgNEhQKDHJlYWR5X2ltYWdlcxgIIAEoBBIbChNyZWFkeV9pbWFnZV9kaWdlc3RzGAkgAygJEhsKE3V0aWxfZGlza190b3RhbF9taWIYCiABKAQSGgoSdXRpbF9kaXNrX3VzZWRfbWliGAsgASgEEhoKEnV0aWxfbWVtX3RvdGFsX21pYhgMIAEoBBIZChF1dGlsX21lbV91c2VkX21pYhgNIAEoBBIUCgx1dGlsX2NwdV9wY3QYDiABKAISGQoRbGFzdF9oZWFydGJlYXRfYXQYDyABKAkSEAoIY29yZG9uZWQYECABKAgSFwoPYWxsb2NhdGFibGVfbWliGBEgASgEEhQKDHJlc2VydmVkX21pYhgSIAEoBBIQCghmcmVlX21pYhgTIAEoBBITCgt0b3RhbF92Y3B1cxgUIAEoDRIYChBjcHVfYnVkZ2V0X3ZjcHVzGBUgASgEEhYKDnJlc2VydmVkX3ZjcHVzGBYgASgEEhIKCmZyZWVfdmNwdXMYFyABKAQSGQoRdXRpbF9iYXNlX3NobV9taWIYGCABKAQSGwoTdXRpbF9wYXJrZWRfcHNzX21pYhgZIAEoBBIcChR1dGlsX3J1bm5pbmdfcHNzX21pYhgaIAEoBBIcChRmYWlsaW5nX2NhcGFiaWxpdGllcxgbIAMoCRIbChNmY19zbmFwc2hvdF92ZXJzaW9uGBwgASgJEhsKE2NhcGFiaWxpdGllc19zY2hlbWEYHSABKA0SGQoRbGl2ZV9tYXRlcmlhbGl6ZXMYHiABKA0SGQoRbGl2ZV9jYXB0dXJlX2pvYnMYHyABKA0SHwoXdXRpbF9jb21taXR0ZWRfc3dhcF9taWIYICABKAQSFAoMY29yZG9uX293bmVyGCEgASgJEjEKCnJldGlyZW1lbnQYIiABKAsyHS5lbmdyYW0uYXBwLnYxLkhvc3RSZXRpcmVtZW50SgQIBxAIUg9sb2NhbF9zbmFwc2hvdHMiIQoOR2V0SG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCSI4Cg9HZXRIb3N0UmVzcG9uc2USJQoEaG9zdBgBIAEoCzIXLmVuZ3JhbS5hcHAudjEuSG9zdFZpZXciKQoWR2V0SG9zdENvd1N0YXRlUmVxdWVzdBIPCgdob3N0X2lkGAEgASgJIlkKF0dldEhvc3RDb3dTdGF0ZVJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSLQoIc2Vzc2lvbnMYAiADKAsyGy5lbmdyYW0uYXBwLnYxLkNvd1N0YXRlVmlldyIjChBEcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkiEwoRRHJhaW5Ib3N0UmVzcG9uc2UiKAoVQWRtaW5EcmFpbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkiXgoWQWRtaW5EcmFpbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEg8KB3BsYW5uZWQYAiADKAkSEQoJZGVzY2VuZGVkGAMgAygJEg8KB3NraXBwZWQYBCABKA0iMwoRQ29yZG9uSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCRINCgVvd25lchgCIAEoCSI8ChdCZWdpbkhvc3RIYW5kb2ZmUmVxdWVzdBIPCgdob3N0X2lkGAEgASgJEhAKCHR0bF9zZWNzGAIgASgEIj0KGEJlZ2luSG9zdEhhbmRvZmZSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEhAKCGFjY2VwdGVkGAIgASgIIjUKEkNvcmRvbkhvc3RSZXNwb25zZRIPCgdob3N0X2lkGAEgASgJEg4KBnN0YXR1cxgCIAEoCSI1ChNVbmNvcmRvbkhvc3RSZXF1ZXN0Eg8KB2hvc3RfaWQYASABKAkSDQoFb3duZXIYAiABKAkiNwoUVW5jb3Jkb25Ib3N0UmVzcG9uc2USDwoHaG9zdF9pZBgBIAEoCRIOCgZzdGF0dXMYAiABKAkiJAoRRGVsZXRlSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCSIUChJEZWxldGVIb3N0UmVzcG9uc2UiGgoYR2V0U3RvcmFnZVN1bW1hcnlSZXF1ZXN0IoQCChlHZXRTdG9yYWdlU3VtbWFyeVJlc3BvbnNlEhEKCXNuYXBzaG90cxgBIAEoBBIWCg5zbmFwc2hvdF9ieXRlcxgCIAEoBBISCgpnY19wZW5kaW5nGAMgASgEEhkKEXRyYWNrZWRfc2FuZGJveGVzGAQgASgEEhQKDGRpcnR5X2NodW5rcxgFIAEoBBIXCg91bmZsdXNoZWRfYnl0ZXMYBiABKAQSGAoQYXZnX2xvY2FsaXR5X3BjdBgHIAEoDRIqCgRyb3dzGAggAygLMhwuZW5ncmFtLmFwcC52MS5EdXJhYmlsaXR5Um93EhgKEGdjX3BlbmRpbmdfZXhhY3QYCSABKAgi5QEKDUR1cmFiaWxpdHlSb3cSEgoKc2FuZGJveF9pZBgBIAEoCRIXCgpzZXNzaW9uX2lkGAIgASgJSACIAQESDwoHaG9zdF9pZBgDIAEoCRIUCgxkaXJ0eV9jaHVua3MYBCABKA0SEwoLZGlydHlfYnl0ZXMYBSABKAQSEwoLYmFzZV9jaHVua3MYBiABKA0SGQoRYmFzZV9jaHVua3NfbG9jYWwYByABKA0SGgoNbGFzdF9mbHVzaF9hdBgIIAEoCUgBiAEBQg0KC19zZXNzaW9uX2lkQhAKDl9sYXN0X2ZsdXNoX2F0IikKE0ZsdXNoU2Vzc2lvblJlcXVlc3QSEgoKc2Vzc2lvbl9pZBgBIAEoCSJbChRGbHVzaFNlc3Npb25SZXNwb25zZRIPCgdvdXRjb21lGAEgASgJEh0KEG1hbmlmZXN0X3ZlcnNpb24YAiABKARIAIgBAUITChFfbWFuaWZlc3RfdmVyc2lvbiJWChZUZWxlcG9ydFNlc3Npb25SZXF1ZXN0EhIKCnNlc3Npb25faWQYASABKAkSGAoLdGFyZ2V0X2hvc3QYAiABKAlIAIgBAUIOCgxfdGFyZ2V0X2hvc3QiUgoXVGVsZXBvcnRTZXNzaW9uUmVzcG9uc2USEwoLdGVsZXBvcnRfaWQYASABKAkSDAoEa2luZBgCIAEoCRIUCgxkZXN0X2hvc3RfaWQYAyABKAkiFwoVR2V0RmxlZXREZW1hbmRSZXF1ZXN0IvEBChZHZXRGbGVldERlbWFuZFJlc3BvbnNlEhMKC3JlYWR5X2hvc3RzGAEgASgNEhkKEXNjaGVkdWxhYmxlX2hvc3RzGAIgASgNEhAKCGZyZWVfbWliGAMgASgEEhEKCXRvdGFsX21pYhgEIAEoBBISCgpmcmVlX3ZjcHVzGAUgASgEEhMKC3RvdGFsX3ZjcHVzGAYgASgEEhYKDmNvcmRvbmVkX2hvc3RzGAcgASgNEhcKD3F1ZXVlZF9zZXNzaW9ucxgIIAEoBBISCgpxdWV1ZWRfbWliGAkgASgEEhQKDHF1ZXVlZF92Y3B1cxgKIAEoBCJJCg5DaHVua0djUmVxdWVzdBIPCgdkcnlfcnVuGAEgASgIEhcKCmdyYWNlX3NlY3MYAiABKARIAIgBAUINCgtfZ3JhY2Vfc2VjcyKMAgoPQ2h1bmtHY1Jlc3BvbnNlEhUKDWxpc3RlZF9jaHVua3MYASABKAQSFgoObWFsZm9ybWVkX2tleXMYAiABKAQSFAoMcGluX3NldF9zaXplGAMgASgEEhkKEWNhbmRpZGF0ZXNfbWFya2VkGAQgASgEEhgKEHByb21vdGVkX2RlbGV0ZXMYByABKAQSHQoVcHJvbW90ZV9kZWxldGVfZXJyb3JzGAggASgEEhIKCmdyYWNlX3NlY3MYCSABKAQSGAoQZ2VuZXJhdGlvbl9tb3ZlZBgKIAEoCBIXCgptYXJrX2Vycm9yGAsgASgJSACIAQFCDQoLX21hcmtfZXJyb3JKBAgFEAZKBAgGEAciSgoPQnVuZGxlR2NSZXF1ZXN0Eg8KB2RyeV9ydW4YASABKAgSFwoKZ3JhY2Vfc2VjcxgCIAEoBEgAiAEBQg0KC19ncmFjZV9zZWNzIsMBChBCdW5kbGVHY1Jlc3BvbnNlEg4KBmxpc3RlZBgBIAEoBBIUCgxwaW5fc2V0X3NpemUYAiABKAQSGQoRY2FuZGlkYXRlc19tYXJrZWQYAyABKAQSGAoQcHJvbW90ZWRfZGVsZXRlcxgEIAEoBBIdChVwcm9tb3RlX2RlbGV0ZV9lcnJvcnMYBSABKAQSFQoNcmVzdGFydF9jb3VudBgGIAEoDRIeChZwcm9tb3RlX3JlcGlubmVkX3NraXBzGAcgASgEIlAKFVNuYXBzaG90QmxvYkdjUmVxdWVzdBIPCgdkcnlfcnVuGAEgASgIEhcKCmdyYWNlX3NlY3MYAiABKARIAIgBAUINCgtfZ3JhY2Vfc2VjcyLcAQoWU25hcHNob3RCbG9iR2NSZXNwb25zZRIOCgZsaXN0ZWQYASABKAQSEQoJbWFsZm9ybWVkGAIgASgEEhQKDHBpbl9zZXRfc2l6ZRgDIAEoBBIZChFjYW5kaWRhdGVzX21hcmtlZBgEIAEoBBIYChBwcm9tb3RlZF9kZWxldGVzGAUgASgEEh4KFnByb21vdGVfcmVwaW5uZWRfc2tpcHMYBiABKAQSHQoVcHJvbW90ZV9kZWxldGVfZXJyb3JzGAcgASgEEhUKDXJlc3RhcnRfY291bnQYCCABKA0iQwoRUmV0aXJlSG9zdFJlcXVlc3QSDwoHaG9zdF9pZBgBIAEoCRINCgVvd25lchgCIAEoCRIOCgZyZWFzb24YAyABKAkicwoSUmV0aXJlSG9zdFJlc3BvbnNlEg8KB2hvc3RfaWQYASABKAkSMQoKcmV0aXJlbWVudBgCIAEoCzIdLmVuZ3JhbS5hcHAudjEuSG9zdFJldGlyZW1lbnQSGQoRdGVsZXBvcnRzX3BsYW5uZWQYAyABKA0ibgoOSG9zdFJldGlyZW1lbnQSFAoMcmVxdWVzdGVkX2F0GAEgASgJEhIKCnJldGlyZWRfYXQYAiABKAkSMgoIYmxvY2tlcnMYAyADKAsyIC5lbmdyYW0uYXBwLnYxLlJldGlyZW1lbnRCbG9ja2VyIjAKEVJldGlyZW1lbnRCbG9ja2VyEgwKBGtpbmQYASABKAkSDQoFY291bnQYAiABKAQy6AsKDEZsZWV0U2VydmljZRJOCglMaXN0SG9zdHMSHy5lbmdyYW0uYXBwLnYxLkxpc3RIb3N0c1JlcXVlc3QaIC5lbmdyYW0uYXBwLnYxLkxpc3RIb3N0c1Jlc3BvbnNlEkgKB0dldEhvc3QSHS5lbmdyYW0uYXBwLnYxLkdldEhvc3RSZXF1ZXN0Gh4uZW5ncmFtLmFwcC52MS5HZXRIb3N0UmVzcG9uc2USUQoKUmV0aXJlSG9zdBIgLmVuZ3JhbS5hcHAudjEuUmV0aXJlSG9zdFJlcXVlc3QaIS5lbmdyYW0uYXBwLnYxLlJldGlyZUhvc3RSZXNwb25zZRJgCg9HZXRIb3N0Q293U3RhdGUSJS5lbmdyYW0uYXBwLnYxLkdldEhvc3RDb3dTdGF0ZVJlcXVlc3QaJi5lbmdyYW0uYXBwLnYxLkdldEhvc3RDb3dTdGF0ZVJlc3BvbnNlEk4KCURyYWluSG9zdBIfLmVuZ3JhbS5hcHAudjEuRHJhaW5Ib3N0UmVxdWVzdBogLmVuZ3JhbS5hcHAudjEuRHJhaW5Ib3N0UmVzcG9uc2USXQoOQWRtaW5EcmFpbkhvc3QSJC5lbmdyYW0uYXBwLnYxLkFkbWluRHJhaW5Ib3N0UmVxdWVzdBolLmVuZ3JhbS5hcHAudjEuQWRtaW5EcmFpbkhvc3RSZXNwb25zZRJRCgpDb3Jkb25Ib3N0EiAuZW5ncmFtLmFwcC52MS5Db3Jkb25Ib3N0UmVxdWVzdBohLmVuZ3JhbS5hcHAudjEuQ29yZG9uSG9zdFJlc3BvbnNlEmMKEEJlZ2luSG9zdEhhbmRvZmYSJi5lbmdyYW0uYXBwLnYxLkJlZ2luSG9zdEhhbmRvZmZSZXF1ZXN0GicuZW5ncmFtLmFwcC52MS5CZWdpbkhvc3RIYW5kb2ZmUmVzcG9uc2USVwoMVW5jb3Jkb25Ib3N0EiIuZW5ncmFtLmFwcC52MS5VbmNvcmRvbkhvc3RSZXF1ZXN0GiMuZW5ncmFtLmFwcC52MS5VbmNvcmRvbkhvc3RSZXNwb25zZRJRCgpEZWxldGVIb3N0EiAuZW5ncmFtLmFwcC52MS5EZWxldGVIb3N0UmVxdWVzdBohLmVuZ3JhbS5hcHAudjEuRGVsZXRlSG9zdFJlc3BvbnNlEmYKEUdldFN0b3JhZ2VTdW1tYXJ5EicuZW5ncmFtLmFwcC52MS5HZXRTdG9yYWdlU3VtbWFyeVJlcXVlc3QaKC5lbmdyYW0uYXBwLnYxLkdldFN0b3JhZ2VTdW1tYXJ5UmVzcG9uc2USVwoMRmx1c2hTZXNzaW9uEiIuZW5ncmFtLmFwcC52MS5GbHVzaFNlc3Npb25SZXF1ZXN0GiMuZW5ncmFtLmFwcC52MS5GbHVzaFNlc3Npb25SZXNwb25zZRJgCg9UZWxlcG9ydFNlc3Npb24SJS5lbmdyYW0uYXBwLnYxLlRlbGVwb3J0U2Vzc2lvblJlcXVlc3QaJi5lbmdyYW0uYXBwLnYxLlRlbGVwb3J0U2Vzc2lvblJlc3BvbnNlEkgKB0NodW5rR2MSHS5lbmdyYW0uYXBwLnYxLkNodW5rR2NSZXF1ZXN0Gh4uZW5ncmFtLmFwcC52MS5DaHVua0djUmVzcG9uc2USSwoIQnVuZGxlR2MSHi5lbmdyYW0uYXBwLnYxLkJ1bmRsZUdjUmVxdWVzdBofLmVuZ3JhbS5hcHAudjEuQnVuZGxlR2NSZXNwb25zZRJdCg5TbmFwc2hvdEJsb2JHYxIkLmVuZ3JhbS5hcHAudjEuU25hcHNob3RCbG9iR2NSZXF1ZXN0GiUuZW5ncmFtLmFwcC52MS5TbmFwc2hvdEJsb2JHY1Jlc3BvbnNlEl0KDkdldEZsZWV0RGVtYW5kEiQuZW5ncmFtLmFwcC52MS5HZXRGbGVldERlbWFuZFJlcXVlc3QaJS5lbmdyYW0uYXBwLnYxLkdldEZsZWV0RGVtYW5kUmVzcG9uc2ViBnByb3RvMw", [file_engram_app_v1_session]); /** * @generated from message engram.app.v1.ListHostsRequest @@ -421,10 +421,7 @@ export const AdminDrainHostRequestSchema: GenMessage = /* messageDesc(file_engram_app_v1_fleet, 9); /** - * Mirrors api/admin.rs DrainHostResponse (renamed: this RPC carries the - * admin cordon+evacuate semantics; the soft status flip above keeps the - * plain DrainHost name). HTTP returns 202 — the evac_resumer scanner is - * the actual deliverable. + * Admission plans return immediately; the teleport machine drives each move. * * @generated from message engram.app.v1.AdminDrainHostResponse */ @@ -435,17 +432,19 @@ export type AdminDrainHostResponse = Message<"engram.app.v1.AdminDrainHostRespon hostId: string; /** - * Session ids now marked Evacuating; the scanner resumes each on a - * peer host within its sweep interval. - * - * @generated from field: repeated string evacuating = 2; + * @generated from field: repeated string planned = 2; + */ + planned: string[]; + + /** + * @generated from field: repeated string descended = 3; */ - evacuating: string[]; + descended: string[]; /** - * @generated from field: repeated engram.app.v1.DrainFailure failures = 3; + * @generated from field: uint32 skipped = 4; */ - failures: DrainFailure[]; + skipped: number; }; /** @@ -455,30 +454,6 @@ export type AdminDrainHostResponse = Message<"engram.app.v1.AdminDrainHostRespon export const AdminDrainHostResponseSchema: GenMessage = /*@__PURE__*/ messageDesc(file_engram_app_v1_fleet, 10); -/** - * @generated from message engram.app.v1.DrainFailure - */ -export type DrainFailure = Message<"engram.app.v1.DrainFailure"> & { - /** - * @generated from field: string session_id = 1; - */ - sessionId: string; - - /** - * Human-readable diagnostic; not machine-parsed, not stable. - * - * @generated from field: string error = 2; - */ - error: string; -}; - -/** - * Describes the message engram.app.v1.DrainFailure. - * Use `create(DrainFailureSchema)` to create a new message. - */ -export const DrainFailureSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 11); - /** * @generated from message engram.app.v1.CordonHostRequest */ @@ -502,7 +477,7 @@ export type CordonHostRequest = Message<"engram.app.v1.CordonHostRequest"> & { * Use `create(CordonHostRequestSchema)` to create a new message. */ export const CordonHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 12); + messageDesc(file_engram_app_v1_fleet, 11); /** * ADR 0116 A-D2: declare a planned handoff — extend the host's @@ -532,7 +507,7 @@ export type BeginHostHandoffRequest = Message<"engram.app.v1.BeginHostHandoffReq * Use `create(BeginHostHandoffRequestSchema)` to create a new message. */ export const BeginHostHandoffRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 13); + messageDesc(file_engram_app_v1_fleet, 12); /** * @generated from message engram.app.v1.BeginHostHandoffResponse @@ -558,7 +533,7 @@ export type BeginHostHandoffResponse = Message<"engram.app.v1.BeginHostHandoffRe * Use `create(BeginHostHandoffResponseSchema)` to create a new message. */ export const BeginHostHandoffResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 14); + messageDesc(file_engram_app_v1_fleet, 13); /** * @generated from message engram.app.v1.CordonHostResponse @@ -582,7 +557,7 @@ export type CordonHostResponse = Message<"engram.app.v1.CordonHostResponse"> & { * Use `create(CordonHostResponseSchema)` to create a new message. */ export const CordonHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 15); + messageDesc(file_engram_app_v1_fleet, 14); /** * @generated from message engram.app.v1.UncordonHostRequest @@ -606,7 +581,7 @@ export type UncordonHostRequest = Message<"engram.app.v1.UncordonHostRequest"> & * Use `create(UncordonHostRequestSchema)` to create a new message. */ export const UncordonHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 16); + messageDesc(file_engram_app_v1_fleet, 15); /** * Same Rust type as CordonHostResponse (CordonResponse); a distinct @@ -633,7 +608,7 @@ export type UncordonHostResponse = Message<"engram.app.v1.UncordonHostResponse"> * Use `create(UncordonHostResponseSchema)` to create a new message. */ export const UncordonHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 17); + messageDesc(file_engram_app_v1_fleet, 16); /** * @generated from message engram.app.v1.DeleteHostRequest @@ -650,7 +625,7 @@ export type DeleteHostRequest = Message<"engram.app.v1.DeleteHostRequest"> & { * Use `create(DeleteHostRequestSchema)` to create a new message. */ export const DeleteHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 18); + messageDesc(file_engram_app_v1_fleet, 17); /** * Mirrors the old `DELETE /api/admin/hosts/:id` (204 No Content) — the @@ -667,7 +642,7 @@ export type DeleteHostResponse = Message<"engram.app.v1.DeleteHostResponse"> & { * Use `create(DeleteHostResponseSchema)` to create a new message. */ export const DeleteHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 19); + messageDesc(file_engram_app_v1_fleet, 18); /** * @generated from message engram.app.v1.GetStorageSummaryRequest @@ -680,7 +655,7 @@ export type GetStorageSummaryRequest = Message<"engram.app.v1.GetStorageSummaryR * Use `create(GetStorageSummaryRequestSchema)` to create a new message. */ export const GetStorageSummaryRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 20); + messageDesc(file_engram_app_v1_fleet, 19); /** * Mirrors api/storage.rs StorageSummaryResponse (ADR 0029): fleet-wide @@ -760,7 +735,7 @@ export type GetStorageSummaryResponse = Message<"engram.app.v1.GetStorageSummary * Use `create(GetStorageSummaryResponseSchema)` to create a new message. */ export const GetStorageSummaryResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 21); + messageDesc(file_engram_app_v1_fleet, 20); /** * One per-sandbox row of the durability ledger. @@ -824,7 +799,7 @@ export type DurabilityRow = Message<"engram.app.v1.DurabilityRow"> & { * Use `create(DurabilityRowSchema)` to create a new message. */ export const DurabilityRowSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 22); + messageDesc(file_engram_app_v1_fleet, 21); /** * @generated from message engram.app.v1.FlushSessionRequest @@ -841,7 +816,7 @@ export type FlushSessionRequest = Message<"engram.app.v1.FlushSessionRequest"> & * Use `create(FlushSessionRequestSchema)` to create a new message. */ export const FlushSessionRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 23); + messageDesc(file_engram_app_v1_fleet, 22); /** * Mirrors api/admin.rs FlushNowResult. @@ -873,64 +848,56 @@ export type FlushSessionResponse = Message<"engram.app.v1.FlushSessionResponse"> * Use `create(FlushSessionResponseSchema)` to create a new message. */ export const FlushSessionResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 24); + messageDesc(file_engram_app_v1_fleet, 23); /** - * @generated from message engram.app.v1.EvacuateSessionRequest + * @generated from message engram.app.v1.TeleportSessionRequest */ -export type EvacuateSessionRequest = Message<"engram.app.v1.EvacuateSessionRequest"> & { +export type TeleportSessionRequest = Message<"engram.app.v1.TeleportSessionRequest"> & { /** * @generated from field: string session_id = 1; */ sessionId: string; /** - * Reserved for a future operator override; carried on today's HTTP - * body (api/admin.rs EvacuateSessionRequest.target_host) but IGNORED - * by the handler — the scanner picks any non-source host via the - * standard policy. - * * @generated from field: optional string target_host = 2; */ targetHost?: string; }; /** - * Describes the message engram.app.v1.EvacuateSessionRequest. - * Use `create(EvacuateSessionRequestSchema)` to create a new message. + * Describes the message engram.app.v1.TeleportSessionRequest. + * Use `create(TeleportSessionRequestSchema)` to create a new message. */ -export const EvacuateSessionRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 25); +export const TeleportSessionRequestSchema: GenMessage = /*@__PURE__*/ + messageDesc(file_engram_app_v1_fleet, 24); /** - * Mirrors api/admin.rs EvacuateSessionResponse. HTTP returns 202; the - * evac_resumer scanner completes the transition. - * - * @generated from message engram.app.v1.EvacuateSessionResponse + * @generated from message engram.app.v1.TeleportSessionResponse */ -export type EvacuateSessionResponse = Message<"engram.app.v1.EvacuateSessionResponse"> & { +export type TeleportSessionResponse = Message<"engram.app.v1.TeleportSessionResponse"> & { /** - * @generated from field: string session_id = 1; + * @generated from field: string teleport_id = 1; */ - sessionId: string; + teleportId: string; /** - * Always "evacuating" on success — the session is paused, - * snapshotted, and the scanner will resume it on a peer within the - * next sweep interval (≤10s default). Watch the session's event - * stream for the Evacuating → Created → Active chain. - * - * @generated from field: string status = 2; + * @generated from field: string kind = 2; */ - status: string; + kind: string; + + /** + * @generated from field: string dest_host_id = 3; + */ + destHostId: string; }; /** - * Describes the message engram.app.v1.EvacuateSessionResponse. - * Use `create(EvacuateSessionResponseSchema)` to create a new message. + * Describes the message engram.app.v1.TeleportSessionResponse. + * Use `create(TeleportSessionResponseSchema)` to create a new message. */ -export const EvacuateSessionResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 26); +export const TeleportSessionResponseSchema: GenMessage = /*@__PURE__*/ + messageDesc(file_engram_app_v1_fleet, 25); /** * @generated from message engram.app.v1.GetFleetDemandRequest @@ -943,7 +910,7 @@ export type GetFleetDemandRequest = Message<"engram.app.v1.GetFleetDemandRequest * Use `create(GetFleetDemandRequestSchema)` to create a new message. */ export const GetFleetDemandRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 27); + messageDesc(file_engram_app_v1_fleet, 26); /** * Mirrors api/admin.rs FleetDemandResponse (ADR 0044 K4 / 0047 / 0048). The @@ -1028,7 +995,7 @@ export type GetFleetDemandResponse = Message<"engram.app.v1.GetFleetDemandRespon * Use `create(GetFleetDemandResponseSchema)` to create a new message. */ export const GetFleetDemandResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 28); + messageDesc(file_engram_app_v1_fleet, 27); /** * @generated from message engram.app.v1.ChunkGcRequest @@ -1065,7 +1032,7 @@ export type ChunkGcRequest = Message<"engram.app.v1.ChunkGcRequest"> & { * Use `create(ChunkGcRequestSchema)` to create a new message. */ export const ChunkGcRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 29); + messageDesc(file_engram_app_v1_fleet, 28); /** * Mirrors api/admin.rs ChunkGcSweepResult. @@ -1144,7 +1111,7 @@ export type ChunkGcResponse = Message<"engram.app.v1.ChunkGcResponse"> & { * Use `create(ChunkGcResponseSchema)` to create a new message. */ export const ChunkGcResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 30); + messageDesc(file_engram_app_v1_fleet, 29); /** * @generated from message engram.app.v1.BundleGcRequest @@ -1169,7 +1136,7 @@ export type BundleGcRequest = Message<"engram.app.v1.BundleGcRequest"> & { * Use `create(BundleGcRequestSchema)` to create a new message. */ export const BundleGcRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 31); + messageDesc(file_engram_app_v1_fleet, 30); /** * Mirrors bundle_gc.rs BundleSweepReport (ADR 0035 §5). Unlike the @@ -1226,7 +1193,7 @@ export type BundleGcResponse = Message<"engram.app.v1.BundleGcResponse"> & { * Use `create(BundleGcResponseSchema)` to create a new message. */ export const BundleGcResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 32); + messageDesc(file_engram_app_v1_fleet, 31); /** * @generated from message engram.app.v1.SnapshotBlobGcRequest @@ -1251,7 +1218,7 @@ export type SnapshotBlobGcRequest = Message<"engram.app.v1.SnapshotBlobGcRequest * Use `create(SnapshotBlobGcRequestSchema)` to create a new message. */ export const SnapshotBlobGcRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 33); + messageDesc(file_engram_app_v1_fleet, 32); /** * Mirrors snapshot_blob_gc.rs SnapshotBlobSweepReport (ADR 0028 @@ -1314,7 +1281,7 @@ export type SnapshotBlobGcResponse = Message<"engram.app.v1.SnapshotBlobGcRespon * Use `create(SnapshotBlobGcResponseSchema)` to create a new message. */ export const SnapshotBlobGcResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 34); + messageDesc(file_engram_app_v1_fleet, 33); /** * ADR 0123 A3: the idempotent retirement request. Cordons the host with @@ -1350,7 +1317,7 @@ export type RetireHostRequest = Message<"engram.app.v1.RetireHostRequest"> & { * Use `create(RetireHostRequestSchema)` to create a new message. */ export const RetireHostRequestSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 35); + messageDesc(file_engram_app_v1_fleet, 34); /** * @generated from message engram.app.v1.RetireHostResponse @@ -1379,7 +1346,7 @@ export type RetireHostResponse = Message<"engram.app.v1.RetireHostResponse"> & { * Use `create(RetireHostResponseSchema)` to create a new message. */ export const RetireHostResponseSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 36); + messageDesc(file_engram_app_v1_fleet, 35); /** * @generated from message engram.app.v1.HostRetirement @@ -1413,7 +1380,7 @@ export type HostRetirement = Message<"engram.app.v1.HostRetirement"> & { * Use `create(HostRetirementSchema)` to create a new message. */ export const HostRetirementSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 37); + messageDesc(file_engram_app_v1_fleet, 36); /** * @generated from message engram.app.v1.RetirementBlocker @@ -1435,7 +1402,7 @@ export type RetirementBlocker = Message<"engram.app.v1.RetirementBlocker"> & { * Use `create(RetirementBlockerSchema)` to create a new message. */ export const RetirementBlockerSchema: GenMessage = /*@__PURE__*/ - messageDesc(file_engram_app_v1_fleet, 38); + messageDesc(file_engram_app_v1_fleet, 37); /** * Fleet, storage, and GC operations (ADR 0051 §2.3, rev 2026-06-10). @@ -1572,14 +1539,15 @@ export const FleetService: GenService<{ output: typeof FlushSessionResponseSchema; }, /** - * ADR 0018 async evacuation (POST /api/admin/sessions/:id/evacuate). + * ADR 0123 B: admit one teleport for a session (synchronous admission; + * the durable machine drives it). Replaces the ADR 0018 evacuate route. * - * @generated from rpc engram.app.v1.FleetService.EvacuateSession + * @generated from rpc engram.app.v1.FleetService.TeleportSession */ - evacuateSession: { + teleportSession: { methodKind: "unary"; - input: typeof EvacuateSessionRequestSchema; - output: typeof EvacuateSessionResponseSchema; + input: typeof TeleportSessionRequestSchema; + output: typeof TeleportSessionResponseSchema; }, /** * The three GC sweeps (ADR 0016 Phase C / ADR 0035 §5 / ADR 0028 diff --git a/web/src/hooks/useTeleportSession.ts b/web/src/hooks/useTeleportSession.ts index 07552c1ba..246228e61 100644 --- a/web/src/hooks/useTeleportSession.ts +++ b/web/src/hooks/useTeleportSession.ts @@ -1,20 +1,8 @@ import { useQueryClient } from "@tanstack/react-query"; import { createConnectQueryKey, useMutation } from "@connectrpc/connect-query"; -import { evacuateSession } from "../gen/engram/app/v1/fleet-FleetService_connectquery"; +import { teleportSession } from "../gen/engram/app/v1/fleet-FleetService_connectquery"; import { getSession } from "../gen/engram/app/v1/session-SessionService_connectquery"; -import type { GetSessionResponse } from "../gen/engram/app/v1/session_pb"; -// ADR 0051 Task 28: migrated from REST POST /api/v1/admin/sessions/:id/teleport -// to FleetService.EvacuateSession (connect-query passthrough via the orchestrator). -// EvacuateSessionRequest accepts sessionId + optional targetHost; the coordinator -// scanner picks the target when targetHost is omitted, but the UI passes the -// admin-chosen host just as the old REST path did. -// -// ADR 0045 Phase F: teleport an Active session to a chosen host. Optimistic: -// the session flips to `evacuating` immediately in the connect-query getSession -// cache so the status rail redraws without waiting on the next poll/event; on -// error we roll back, and we always invalidate on settle so the SSE/poll view is -// authoritative as the `evacuating → created → active` chain lands on the target. export function useTeleportSession(sessionId: string) { const qc = useQueryClient(); const sessionKey = createConnectQueryKey({ @@ -22,27 +10,11 @@ export function useTeleportSession(sessionId: string) { input: { sessionId }, cardinality: "finite", }); - // evacuateSession is FleetService.EvacuateSession — admin-gated at the - // orchestrator CASL layer. The input shape takes sessionId + optional targetHost. - const mutation = useMutation(evacuateSession, { - onMutate: async () => { - await qc.cancelQueries({ queryKey: sessionKey }); - const previous = qc.getQueryData(sessionKey); - qc.setQueryData(sessionKey, (old) => - old?.session ? { ...old, session: { ...old.session, status: "evacuating" } } : old, - ); - return { previous }; - }, - onError: (_err, _input, ctx) => { - const rollback = (ctx as { previous?: GetSessionResponse } | undefined)?.previous; - if (rollback) qc.setQueryData(sessionKey, rollback); - }, + const mutation = useMutation(teleportSession, { onSettled: () => { qc.invalidateQueries({ queryKey: sessionKey }); }, }); - - // Expose the same interface as before: mutate(targetHostId). return { mutate: (targetHostId: string) => mutation.mutate({ sessionId, targetHost: targetHostId }), mutateAsync: (targetHostId: string) => @@ -50,5 +22,6 @@ export function useTeleportSession(sessionId: string) { isPending: mutation.isPending, error: mutation.error, status: mutation.status, + data: mutation.data, }; } diff --git a/web/src/lib/types.ts b/web/src/lib/types.ts index 9898b58dd..248f1b6e9 100644 --- a/web/src/lib/types.ts +++ b/web/src/lib/types.ts @@ -39,7 +39,7 @@ * reconciler resolves this to `idle` (if a * recoverable snapshot exists) or `dead`. * evacuating — mid-relocation to a peer host (ADR 0018); the - * evac_resumer scanner drives it back to active. + * teleport driver drives it back to active. * evicting — idle-eviction in flight (ADR 0034); the eviction * scanner snapshots + suspends it to `idle` within * a couple of minutes. /prompt and /resume 409 diff --git a/web/src/sse.ts b/web/src/sse.ts index ab6211534..f3dae4753 100644 --- a/web/src/sse.ts +++ b/web/src/sse.ts @@ -33,6 +33,7 @@ export interface SseHandlers { /** Durable event discriminants this client subscribes to explicitly. */ export const SESSION_EVENT_KINDS: readonly SessionEventKind[] = [ "status_changed", + "teleport_finished", "exec_started", "exec_completed", "stdout", diff --git a/web/src/test-utils.tsx b/web/src/test-utils.tsx index f249160cf..7a4236396 100644 --- a/web/src/test-utils.tsx +++ b/web/src/test-utils.tsx @@ -71,7 +71,7 @@ export const testTransport: Transport = createRouterTransport((router) => { getHost: () => ({ host: undefined }), getHostCowState: () => ({ hostId: "", sessions: [] }), drainHost: () => ({}), - adminDrainHost: () => ({ hostId: "", evacuating: [], failures: [] }), + adminDrainHost: () => ({ hostId: "", planned: [], descended: [], skipped: 0 }), cordonHost: () => ({ hostId: "", status: "" }), uncordonHost: () => ({ hostId: "", status: "" }), getStorageSummary: () => ({ @@ -85,7 +85,7 @@ export const testTransport: Transport = createRouterTransport((router) => { rows: [], }), flushSession: () => ({ outcome: "", manifestVersion: undefined }), - evacuateSession: () => ({ sessionId: "", status: "" }), + teleportSession: () => ({ teleportId: "", kind: "snapshot", destHostId: "" }), chunkGc: () => ({ listedChunks: 0n, malformedKeys: 0n,