From cca34c918e41a717e17edc63c3d0e21c75c1a392 Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Sat, 10 Oct 2026 18:19:28 +0000 Subject: [PATCH 1/6] Restore retained checkpoints across compatible nodes --- .../agents-api/node-generation-protocol.md | 4 +- contracts/agents-api/sandbox-deployment.md | 2 +- .../agents-api/zh/node-generation-protocol.md | 6 +- contracts/agents-api/zh/sandbox-deployment.md | 4 +- deploy/node/node_generations.py | 4 + deploy/node/node_install.py | 56 ++- deploy/node/test_node_generations.py | 2 +- deploy/node/test_node_install.py | 31 +- docs/getting-started/nodes.md | 7 +- docs/getting-started/operations.md | 2 + docs/sandbox-provider.md | 16 +- docs/zh/getting-started/nodes.md | 9 +- docs/zh/getting-started/operations.md | 4 +- docs/zh/sandbox-provider.md | 18 +- services/core/cmd/sandbox-node/main.go | 8 +- services/core/cmd/server/managed_nodes.go | 2 +- .../internal/db/queries/node_generations.sql | 16 +- .../db/queries/runtime_suspension.sql | 21 + .../queries/runtime_suspension_pressure.sql | 19 +- services/core/internal/db/sqlc/models.go | 1 + .../internal/db/sqlc/node_generations.sql.go | 44 +- .../db/sqlc/runtime_suspension.sql.go | 64 +++ .../sqlc/runtime_suspension_pressure.sql.go | 37 +- .../core/internal/deployment/allocations.go | 81 ++- .../internal/deployment/allocations_test.go | 45 +- services/core/internal/deployment/nodes.go | 22 +- .../deployment/placement/placement.go | 54 +- .../deployment/placement/placement_test.go | 20 +- services/core/internal/deployment/storage.go | 10 +- .../deployment/suspension_pressure.go | 49 +- .../internal/execution/runtime_compute.go | 46 +- .../execution/runtime_compute_wake.go | 16 +- .../execution/runtime_observation_test.go | 2 +- .../execution/runtime_replacement_test.go | 10 +- .../execution/runtime_restore_attempt_test.go | 227 +++++++++ .../postgres/deploymentpg/allocations.go | 27 +- .../deploymentpg/checkpoint_transfer_test.go | 301 +++++++++++ .../postgres/deploymentpg/fixture_test.go | 23 +- .../deploymentpg/host_history_test.go | 12 +- .../postgres/deploymentpg/presence_test.go | 8 +- .../deploymentpg/suspension_pressure.go | 30 +- .../deploymentpg/suspension_pressure_test.go | 7 +- .../persistence/postgres/deploymentpg/tx.go | 9 +- .../postgres/placementpg/placementpg.go | 41 +- services/core/internal/sandbox/docker/node.go | 4 +- services/core/internal/sandbox/generation.go | 9 +- .../sandbox/microsandbox/checkpoint_linux.go | 133 +++++ .../microsandbox/checkpoint_linux_test.go | 88 ++++ .../sandbox/microsandbox/checkpoint_other.go | 10 + .../internal/sandbox/microsandbox/identity.go | 11 +- .../internal/sandbox/microsandbox/node.go | 38 +- .../sandbox/microsandbox/node_test.go | 10 +- .../internal/sandbox/microsandbox/provider.go | 7 + .../sandbox/microsandbox/provider_test.go | 4 +- .../microsandbox/restore_settlement_test.go | 41 ++ .../internal/sandbox/microsandbox/types.go | 3 +- .../node/checkpoint_generation_test.go | 98 ++++ .../sandbox/node/generation_json_test.go | 7 +- .../internal/sandbox/node/generation_wire.go | 6 +- .../core/internal/sandbox/node/generations.go | 12 +- .../internal/sandbox/node/generations_test.go | 12 +- services/core/internal/sandbox/node/hub.go | 6 +- services/core/internal/sandbox/node/proxy.go | 6 + services/core/internal/sandbox/node/wire.go | 2 +- .../internal/sandbox/node/workspace_test.go | 2 +- .../core/internal/sandbox/sandbox_provider.go | 34 +- .../000099_checkpoint_compatibility.sql | 6 + .../integration/deployment_fixture_test.go | 15 + .../runtime_compute_lifecycle_test.go | 10 +- .../runtime_node_generations_test.go | 2 +- .../runtime_node_lifecycle_fixture_test.go | 6 +- .../tests/integration/runtime_nodes_test.go | 20 +- .../integration/runtime_suspension_test.go | 13 +- .../sandbox_deployment_worker_test.go | 2 +- .../tools/microsandbox-provider/README.md | 20 +- .../tools/microsandbox-provider/archive.go | 467 ++++++++++++++++++ .../tools/microsandbox-provider/backend.go | 25 +- .../tools/microsandbox-provider/lock_test.go | 23 +- .../core/tools/microsandbox-provider/main.go | 15 +- .../tools/microsandbox-provider/snapshot.go | 144 ++++-- .../snapshot_observation_test.go | 297 ++++++++--- 81 files changed, 2649 insertions(+), 376 deletions(-) create mode 100644 services/core/internal/execution/runtime_restore_attempt_test.go create mode 100644 services/core/internal/persistence/postgres/deploymentpg/checkpoint_transfer_test.go create mode 100644 services/core/internal/sandbox/microsandbox/checkpoint_linux.go create mode 100644 services/core/internal/sandbox/microsandbox/checkpoint_linux_test.go create mode 100644 services/core/internal/sandbox/microsandbox/checkpoint_other.go create mode 100644 services/core/internal/sandbox/microsandbox/restore_settlement_test.go create mode 100644 services/core/internal/sandbox/node/checkpoint_generation_test.go create mode 100644 services/core/migrations/000099_checkpoint_compatibility.sql create mode 100644 services/core/tools/microsandbox-provider/archive.go diff --git a/contracts/agents-api/node-generation-protocol.md b/contracts/agents-api/node-generation-protocol.md index de65b9069..3b19e7fe6 100644 --- a/contracts/agents-api/node-generation-protocol.md +++ b/contracts/agents-api/node-generation-protocol.md @@ -73,9 +73,9 @@ Node startup and generation loading validate complete Provider operation declara ## Generation control -A node without generation management serves only its enrolled generation, with fixed configuration, and receives no preparation or retention frames. A generation-managing node prepares the target generation Core announces in `welcome` and `heartbeat_ack` and keeps serving its durable serving generation while it does; target preparation is independent of the serving provider's readiness. +A node without generation management serves only its enrolled generation, with fixed configuration, and receives no preparation or retention frames. It can report that exact generation in `Health.Generations`; a checkpoint-capable provider must do so, including its verified checkpoint qualification. `provider_ready` and the reported state must agree. A generation-managing node prepares the target generation Core announces in `welcome` and `heartbeat_ack` and keeps serving its durable serving generation while it does; target preparation is independent of the serving provider's readiness. -Its `hello` and heartbeats carry at most eight generation observations. Each names a positive signed-64-bit generation, its lowercase SHA-256 specification digest, a `ready`, `preparing` or `failed` state and an optional fixed diagnostic. The target and serving generations come first; other records rotate fairly. Eight bounds one message, not the number of generations a node may keep. An omitted observation never authorizes deletion or implies absence. +Its `hello` and heartbeats carry at most eight generation observations. Each names a positive signed-64-bit generation, its lowercase SHA-256 specification digest, a `ready`, `preparing` or `failed` state and an optional fixed diagnostic. The target and serving generations come first; other records rotate fairly. Eight bounds one message, not the number of generations a node may keep. An omitted observation never authorizes deletion or implies absence. A ready checkpoint-capable generation includes `checkpoint` with nonempty opaque `artifact_domain` and `execution_class` tokens, each at most 256 bytes. Unsupported providers omit it; preparing or failed generations never advertise it. Core persists the declaration with the authenticated connection and generation, and uses it only while that generation is ready on the current online connection. The [Provider contract](../../docs/sandbox-provider.md#checkpoint-transfer) defines archive and execution compatibility. Retention uses its own bounded exchange. A `retention` request names at most eight local `(generation, specification_digest)` references, a UUID, a sequence that increases by one, the current `connection_id` and the owner epoch. The `retention_ack` must match the complete pending request, entry order and identity included, and give an explicit boolean `keep` for every entry. One exchange is pending per connection, and a disconnect discards it. An unsolicited, replayed, stale, partial or mixed acknowledgement deletes nothing. Retention traffic never uses the Provider request queue. diff --git a/contracts/agents-api/sandbox-deployment.md b/contracts/agents-api/sandbox-deployment.md index b3b79721e..41668d90e 100644 --- a/contracts/agents-api/sandbox-deployment.md +++ b/contracts/agents-api/sandbox-deployment.md @@ -142,7 +142,7 @@ A node is online while it is connected under the current owner epoch with a hear Each node adds `rollout: {state, ready_generation, diagnostic?}`, where `ready_generation` is the nullable durable serving pin and `diagnostic` a fixed code for the target generation; allocation items add `deployment_generation`. Poll every five seconds only while `rollout.state` is `preparing` or `reset` is not null; old Sessions and failed, update-required or offline nodes alone do not keep polling active. -Node-backed creation validates the target deployment and commits the Session, pending Environment and any initial input without reserving compute. A full, offline or preparing fleet leaves that accepted work waiting; an unconfigured deployment, reset or unsupported combination still rejects admission. The common scheduler uses bounded rotating scans of unplaced demand, with pages ordered by recorded time and Environment ID and a fixed time boundary for each scan so continuous arrivals cannot prevent it from revisiting older work. It checks online presence, exact serving-generation readiness, address, shared capacity and the selected generation's Harness/filesystem compatibility, then prefers the newest qualifying pin. Placement is immutable once reserved. A bounded scan skips temporarily unavailable demand and resumes after a restart. Existing suspended allocations restore on their original node through the same capacity lock; this is not a global fairness guarantee across hot restores and unplaced work. Input retains its [original five-minute deadline](./environments.md#reservations), including time waiting for capacity. +Node-backed creation validates the target deployment and commits the Session, pending Environment and any initial input without reserving compute. A full, offline or preparing fleet leaves that accepted work waiting; an unconfigured deployment, reset or unsupported combination still rejects admission. The common scheduler uses bounded rotating scans of unplaced demand, with pages ordered by recorded time and Environment ID and a fixed time boundary for each scan so continuous arrivals cannot prevent it from revisiting older work. It checks online presence, exact serving-generation readiness, address, shared capacity and the selected generation's Harness/filesystem compatibility, then prefers the newest qualifying pin. Placement is immutable once reserved until confirmed release or a settled checkpoint transfer. A bounded scan skips temporarily unavailable demand and resumes after a restart. Existing suspended allocations reserve eligible checkpoint capacity through the same lock under the [checkpoint transfer contract](../../docs/sandbox-provider.md#checkpoint-transfer); this is not a global fairness guarantee across hot restores and unplaced work. Input retains its [original five-minute deadline](./environments.md#reservations), including time waiting for capacity. ## Reset diff --git a/contracts/agents-api/zh/node-generation-protocol.md b/contracts/agents-api/zh/node-generation-protocol.md index 481dbc4af..b035ab7ab 100644 --- a/contracts/agents-api/zh/node-generation-protocol.md +++ b/contracts/agents-api/zh/node-generation-protocol.md @@ -1,7 +1,7 @@ --- title: "沙箱节点协议" source: contracts/agents-api/node-generation-protocol.md -source_hash: 3fa1d9257acefa2d144f6e76ac5c2d3ff1d074ebb2b21fb48a2f79ef58ad6073 +source_hash: fb446e91c3df3c9e397e5e9abb794245b0ae11d81176879207718680e8ab4173 --- 沙箱节点在其主机上运行 Docker 或 microsandbox Provider,并通过一个 WebSocket 与 Core 相连。Core 通过该连接发送 Provider 操作;节点针对本地 Provider 执行这些操作,并报告就绪状态、主机测量值及其持有的部署代次。Core 始终是唯一的生命周期所有者:节点绝不重试变更操作或调度工作。帧和校验器位于 [`services/core/internal/sandbox/node`](https://github.com/MiniMax-AI/OpenAgentCore/tree/main/services/core/internal/sandbox/node)(`wire.go`、`generation_wire.go`);节点用于注册和读取配置的 HTTP 路由位于[机器连接 API](machine-api.md#node-routes)。 @@ -75,9 +75,9 @@ Core 发送包含以下内容的 `request` 帧: ## 代次控制 {#generation-control} -未启用代次管理的节点只服务其登记的代次,配置固定,并且不会收到准备或保留帧。支持代次管理的节点会准备 Core 在 `welcome` 和 `heartbeat_ack` 中通告的目标代次,并在此期间继续服务其持久化的服务代次;目标代次的准备独立于服务 Provider 的就绪状态。 +未启用代次管理的节点只服务其登记的代次,配置固定,并且不会收到准备或保留帧。它可以在 `Health.Generations` 报告该精确代次;支持检查点的 Provider 必须报告,并包含已验证的检查点资格。`provider_ready` 必须与报告的状态一致。支持代次管理的节点会准备 Core 在 `welcome` 和 `heartbeat_ack` 中通告的目标代次,并在此期间继续服务其持久化的服务代次;目标代次的准备独立于服务 Provider 的就绪状态。 -节点的 `hello` 和心跳最多携带八条代次观察记录。每条记录指定一个正值的有符号 64 位代次编号、其小写 SHA-256 规范摘要、`ready`、`preparing` 或 `failed` 状态,以及可选的固定诊断信息。目标代次和服务代次的记录排在前面,其余记录公平轮换。八条记录限制的是单条消息,而不是节点可保留的代次数量。省略某条观察记录绝不会授权删除,也不会暗示不存在。 +节点的 `hello` 和心跳最多携带八条代次观察记录。每条记录指定一个正值的有符号 64 位代次编号、其小写 SHA-256 规范摘要、`ready`、`preparing` 或 `failed` 状态,以及可选的固定诊断信息。目标代次和服务代次的记录排在前面,其余记录公平轮换。八条记录限制的是单条消息,而不是节点可保留的代次数量。省略某条观察记录绝不会授权删除,也不会暗示不存在。 支持检查点的 ready 代次携带 `checkpoint`,包含非空且不透明的 `artifact_domain` 和 `execution_class` token,各最多 256 字节。不支持的 Provider 省略该字段;preparing 或 failed 代次不声明它。Core 将声明与已认证连接和代次一起持久化,仅在当前在线连接的该代次仍 ready 时使用。[Provider 契约](../../../docs/zh/sandbox-provider.md#checkpoint-transfer) 定义归档与执行兼容性。 保留使用独立且有界的交换。`retention` 请求最多指定八个本地 `(generation, specification_digest)` 引用、一个 UUID、一个每次递增 1 的 `sequence`、当前 `connection_id` 和所有者 epoch。`retention_ack` 必须与完整的待处理请求匹配,包括条目顺序和身份信息,并为每个条目给出显式布尔值 `keep`。每条连接只能有一个交换处于待处理状态,断连会将其丢弃。任何未经请求、重放、过期、不完整或混合的确认都不会删除任何内容。保留流量从不使用 Provider 请求队列。 diff --git a/contracts/agents-api/zh/sandbox-deployment.md b/contracts/agents-api/zh/sandbox-deployment.md index db6d81c6c..05fbe1fc8 100644 --- a/contracts/agents-api/zh/sandbox-deployment.md +++ b/contracts/agents-api/zh/sandbox-deployment.md @@ -1,7 +1,7 @@ --- title: "沙箱部署" source: contracts/agents-api/sandbox-deployment.md -source_hash: 245f1791cf2e42b3a40c91aa0df4f390a44709db5e0ffbcae1e12dc4170344fe +source_hash: 537ae957a77d625b3adb7828156352a661ae09270b57b174406c6db159a957bd --- 沙箱部署为 Core 管理的 `openai_hosted` 执行选择 Sandbox Provider、每个沙箱的资源以及不可变的 Runtime 发行版。PostgreSQL 为每个安装维护一个当前有效选择;Web 和 Core API 写入同一配置。节点文件保存其已安装副本和特定于主机的路径,且不能覆盖其资源或 Runtime。该选择独立于 Harness。部署可以保持未配置状态,没有节点;此时它拒绝托管准入。 @@ -145,7 +145,7 @@ POST 会在持久保存候选配置之前对其进行验证,并且不会创建 每个节点会添加 `rollout: {state, ready_generation, diagnostic?}`,其中 `ready_generation` 是可为 null 的持久服务固定状态,`diagnostic` 是目标代次的固定代码;分配项会添加 `deployment_generation`。仅当 `rollout.state` 为 `preparing` 或 `reset` 非 null 时,才每五秒轮询一次;旧 Session 以及失败、需要更新或离线的节点本身均不会使轮询保持活动状态。 -节点模式创建先验证目标部署,提交 Session、pending Environment 及初始输入,但不预留计算资源。节点全部已满、离线或准备中时,已接受的工作继续等待;未配置部署、reset 或不支持的组合仍拒绝准入。共同调度器对尚未放置的需求进行有界循环扫描,分页按需求记录时间和 Environment ID 排序,每轮使用固定时间上界,避免持续的新需求阻止重新访问较早的工作。它检查在线状态、精确服务代次就绪情况、地址、共享容量和所选代次的 Harness/文件系统兼容性,然后优先选择最新的合格固定代次。预留后的 placement 保持不可变。有界扫描跳过暂时不可调度的需求,并在重启后继续。已有 suspended allocation 通过同一容量锁在原节点恢复;这不构成热恢复和未放置工作之间的全局公平性保证。输入保留[原始五分钟期限](./environments.md#reservations),等待容量的时间也计入其中。 +节点模式创建先验证目标部署,提交 Session、pending Environment 及初始输入,但不预留计算资源。节点全部已满、离线或准备中时,已接受的工作继续等待;未配置部署、reset 或不支持的组合仍拒绝准入。共同调度器对尚未放置的需求进行有界循环扫描,分页按需求记录时间和 Environment ID 排序,每轮使用固定时间上界,避免持续的新需求阻止重新访问较早的工作。它检查在线状态、精确服务代次就绪情况、地址、共享容量和所选代次的 Harness/文件系统兼容性,然后优先选择最新的合格固定代次。预留后的 placement 保持不可变,直到确认释放或已结清的检查点转移。有界扫描跳过暂时不可调度的需求,并在重启后继续。已有 suspended allocation 按[检查点转移契约](../../../docs/zh/sandbox-provider.md#checkpoint-transfer),通过同一容量锁预留符合条件的检查点容量;这不构成热恢复和未放置工作之间的全局公平性保证。输入保留[原始五分钟期限](./environments.md#reservations),等待容量的时间也计入其中。 ## 重置 {#reset} diff --git a/deploy/node/node_generations.py b/deploy/node/node_generations.py index 70e61fe0d..d056421f0 100644 --- a/deploy/node/node_generations.py +++ b/deploy/node/node_generations.py @@ -239,6 +239,9 @@ def validate_preparation_plan(root, plan, base, installer): if not any(paths == [release / name for name in installer.MICRO] for release in (root, root / "releases" / source)): raise installer.InstallError("Preparation artifacts are outside their immutable release") expected_home = generation_home(root, plan, base, installer) + if micro["checkpoint_root"] != base["native"]["checkpoint_root"]: + raise installer.InstallError("Preparation checkpoint store differs") + installer.prepare_checkpoint_root(Path(micro["checkpoint_root"]), initialize=False) if Path(micro["runtime_home"]) != expected_home: raise installer.InstallError("Preparation native store differs") for path in paths + [expected_home]: @@ -399,6 +402,7 @@ def prepare(args, installer): raise installer.RuntimeDownloadError("Runtime release provenance differs") from error if args.provider == "microsandbox": args.runtime_home = Path(value["native"]["runtime_home"]) if value else generation_home(root, args.configuration, base, installer) + args.checkpoint_root = Path(base["native"]["checkpoint_root"]) if value is None: value = installer.provider_config(root / "releases" / runtime["source_commit"], args, runtime["image_id"]) if preparation is None: diff --git a/deploy/node/node_install.py b/deploy/node/node_install.py index d563a29cf..5cda1e2fc 100644 --- a/deploy/node/node_install.py +++ b/deploy/node/node_install.py @@ -279,6 +279,49 @@ def micro_home(installation_id): return directory +def checkpoint_root(args): + return getattr(args, "checkpoint_root", None) or Path.home() / ".oac/checkpoints" / hashlib.sha256(args.installation_id.encode()).hexdigest()[:12] + + +def prepare_checkpoint_root(directory, *, initialize=True): + directory = Path(directory) + if not directory.is_absolute() or directory.resolve() != directory: + raise InstallError("Checkpoint root must be a canonical absolute directory") + if not initialize and not directory.is_dir(): + raise InstallError("Retained checkpoint root is missing") + safe_directory(directory) + marker = directory / ".oac-checkpoint-store" + if not existing_file(marker): + if not initialize: + raise InstallError("Retained checkpoint store identity is missing") + if any(directory.iterdir()): + raise InstallError("Checkpoint root contains unowned state") + try: + descriptor = os.open(marker, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600) + except FileExistsError: + pass + else: + with os.fdopen(descriptor, "w") as output: + output.write(str(uuid.uuid4()) + "\n") + output.flush() + os.fsync(output.fileno()) + existing_file(marker) + value = marker.read_text() + try: + identity = uuid.UUID(value.rstrip("\n")) + except ValueError: + raise InstallError("Invalid checkpoint store identity") from None + if identity.int == 0 or value != str(identity) + "\n": + raise InstallError("Invalid checkpoint store identity") + with marker.open("rb") as source: + os.fsync(source.fileno()) + descriptor = os.open(directory, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + + def provider_config(root, args, runtime_image): result = {"installation_id": args.installation_id, "provider": args.provider, "core_url": args.core_url + "/api/v1", "specification": args.configuration["specification"], "generation": args.configuration["generation"]} @@ -294,6 +337,7 @@ def provider_config(root, args, runtime_image): result["native"] = { "helper_path": str(root / MICRO[0]), "runtime_path": str(root / MICRO[1]), "firmware_path": str(root / MICRO[2]), "runtime_home": str(getattr(args, "runtime_home", micro_home(args.installation_id))), + "checkpoint_root": str(checkpoint_root(args)), "network": {"default_egress": "deny", "default_ingress": "deny", "rules": core_rules + [ {"action": "allow", "direction": "egress", "destination": "public"}, {"action": "allow", "direction": "egress", "destination": "host", "protocol": "udp", "port": "53"}, @@ -348,8 +392,17 @@ def configure_node(root, args, token): args.configuration = node_spec.fetch(args, token, retained, open_request, allow_enrollment=not (root / "registered.json").exists()) args.provider = args.configuration["provider"] preflight(args.provider) + if args.provider != "microsandbox" and getattr(args, "checkpoint_root", None) is not None: + raise InstallError("Checkpoint root applies only to microsandbox nodes") if args.provider == "microsandbox": + retained_provider = private_json(root / "provider.json") + if retained_provider: + retained_root = Path(retained_provider["native"]["checkpoint_root"]) + if getattr(args, "checkpoint_root", None) not in (None, retained_root): + raise InstallError("Retained checkpoint root differs; preserve its state") + args.checkpoint_root = retained_root runtime_home = micro_home(args.installation_id) + prepare_checkpoint_root(checkpoint_root(args), initialize=retained_provider is None) safe_directory(runtime_home) owner = runtime_home / "oac-installation.json" if not owner.exists() and any(runtime_home.iterdir()): @@ -1244,6 +1297,7 @@ def main(argv=None): parser.add_argument("--core-url", type=origin) parser.add_argument("--provider", choices=("docker", "microsandbox"), help="Optional assertion; Core owns provider selection") parser.add_argument("--installation-id", required=True) + parser.add_argument("--checkpoint-root", type=Path, help="Private microsandbox checkpoint directory, shared across compatible nodes when configured") parser.add_argument("--enrollment-token-stdin", action="store_true", help="Read the one-time enrollment token from standard input") parser.add_argument("--generation-action", choices=("prepare", "collect"), help=argparse.SUPPRESS) parser.add_argument("--generation", type=int, help=argparse.SUPPRESS) @@ -1272,7 +1326,7 @@ def main(argv=None): if os.geteuid() != 0: raise InstallError("Node installation and removal require root. Run this command with sudo.") if args.uninstall: - if args.source_url or args.bundle or args.core_url or args.provider or args.enrollment_token_stdin: + if args.source_url or args.bundle or args.core_url or args.provider or args.enrollment_token_stdin or args.checkpoint_root: parser.error("--uninstall takes only --installation-id and --force") uninstall_system(args) return diff --git a/deploy/node/test_node_generations.py b/deploy/node/test_node_generations.py index c83ff6f0d..0b2c66311 100644 --- a/deploy/node/test_node_generations.py +++ b/deploy/node/test_node_generations.py @@ -174,7 +174,7 @@ def micro_fixture(self): runtime.chmod(0o700) image = self.value["specification"]["runtime"]["microsandbox_ref"] self.value["specification"]["runtime"]["runtime_sha256"] = hashlib.sha256(runtime.read_bytes()).hexdigest() - self.value["native"] = {"helper_path": str(self.release / "helper"), "runtime_home": str(home), "runtime_path": str(runtime), "firmware_path": str(self.release / "firmware")} + self.value["native"] = {"helper_path": str(self.release / "helper"), "runtime_home": str(home), "checkpoint_root": str(self.root / "checkpoints"), "runtime_path": str(runtime), "firmware_path": str(self.release / "firmware")} self.args.specification_digest = node_spec.digest("microsandbox", self.value["specification"]) node_generations.atomic_json(self.root / "provider.json", self.value) # This fixture changes provider before any helper exists. diff --git a/deploy/node/test_node_install.py b/deploy/node/test_node_install.py index 203db53bf..aada90152 100644 --- a/deploy/node/test_node_install.py +++ b/deploy/node/test_node_install.py @@ -485,7 +485,7 @@ def test_microsandbox_imports_image_and_allows_only_explicit_private_core_endpoi self.install() config = json.loads((self.root / "provider.json").read_text())["native"] # Resources, the image and artifact hashes are read from the specification. - self.assertEqual(set(config), {"helper_path", "runtime_path", "firmware_path", "runtime_home", "network"}) + self.assertEqual(set(config), {"helper_path", "runtime_path", "firmware_path", "runtime_home", "checkpoint_root", "network"}) rules = config["network"]["rules"] self.assertIn({"action": "allow", "direction": "egress", "destination": "172.29.144.1", "protocol": "tcp", "port": "24443"}, rules) self.assertNotIn("private", [rule["destination"] for rule in rules]) @@ -1116,6 +1116,35 @@ def test_microsandbox_short_home_is_stable_and_rejects_long_user_home(self): installer.micro_home("94be54a1-138c-4f30-bc87-b13686272dbe") +class CheckpointRootTests(unittest.TestCase): + def test_identity_is_private_stable_and_not_recreated(self): + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) / "checkpoints" + installer.prepare_checkpoint_root(root) + marker = root / ".oac-checkpoint-store" + first = marker.read_bytes() + installer.prepare_checkpoint_root(root) + self.assertEqual(first, marker.read_bytes()) + self.assertEqual(stat.S_IMODE(root.stat().st_mode), 0o700) + self.assertEqual(stat.S_IMODE(marker.stat().st_mode), 0o600) + marker.write_text("incomplete") + with self.assertRaises(installer.InstallError): + installer.prepare_checkpoint_root(root) + + def test_foreign_contents_and_symlink_are_not_adopted(self): + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) / "checkpoints" + root.mkdir() + (root / "unowned").write_bytes(b"data") + with self.assertRaisesRegex(installer.InstallError, "unowned"): + installer.prepare_checkpoint_root(root) + (root / "unowned").unlink() + marker = root / ".oac-checkpoint-store" + marker.symlink_to(Path(temporary) / "missing") + with self.assertRaises(installer.InstallError): + installer.prepare_checkpoint_root(root) + + class UnsupportedNodeUpdateTests(unittest.TestCase): def test_update_refuses_without_host_operations(self): with self.assertRaisesRegex(installer.InstallError, "not supported;.*reinstall"): diff --git a/docs/getting-started/nodes.md b/docs/getting-started/nodes.md index d4890f614..e35850ca5 100644 --- a/docs/getting-started/nodes.md +++ b/docs/getting-started/nodes.md @@ -12,6 +12,8 @@ You add a node by generating a command in Web and running it on the host. The [s - **The sandbox configuration is saved.** Open **System** → **Manage sandbox configuration**, choose **Own machines**, the backend and a sandbox size, and **Save configuration**. To change a saved configuration, choose **Reset deployment** first. Every node of an installation uses that backend. - **The console can serve the node files.** Nodes download their Runtime and provider files from the console, which redirects to the release for files it does not hold, and check each file's size and SHA-256 against the release manifest. Node hosts therefore need access to the release. Without the files, Add node says *This console has no node files for …*. +For microsandbox checkpoint recovery across nodes, add `--checkpoint-root /absolute/private/shared-store` to the installer command on each participating node. Mount the same private store at those paths before installation and grant only the node service account access. The default is a private node-local store and supports recovery on that node. The checkpoint store is separate from workspace storage and must never be guest-accessible; the [checkpoint transfer contract](../sandbox-provider.md#checkpoint-transfer) owns compatibility and retention requirements. + For microsandbox with independent workspace storage, complete the [NFS mount and service-account setup](../configuration.md#independent-workspace-storage) on this host before enrollment. The node receives the selected immutable filesystem configuration with each binding; do not author a separate node storage setting. The Core host joins like any other host: to run sandboxes on it, add it as a node. @@ -127,7 +129,8 @@ Use manual registration when you manage the node's files and service yourself in 3. Read the node configuration with the token, which does not consume it: `GET /api/v1/sandbox-node/configuration` with `Authorization: Bearer `. 4. Write a private provider file. Copy `provider`, `installation_id`, `core_url`, `generation` and `specification` from the response, and add a `native` object with the host settings of that provider. The adapter reads sandbox size, the Runtime image and artifact hashes from `specification`: - Docker: the [Docker node configuration](../configuration.md#docker-node-configuration) fields, with `host` an explicit Unix socket, `image` the local ID of the imported Runtime image and `seccomp_file` absolute. - - microsandbox: absolute `helper_path`, `runtime_path` and `firmware_path`, a `network` policy, and `runtime_home`: a private directory, which the helper creates with mode `0700` when it is missing. microsandbox places Unix sockets under it, so keep its path within 48 bytes; the installer refuses a longer one for its own nodes. + - microsandbox: absolute `helper_path`, `runtime_path` and `firmware_path`, a `network` policy, an explicit `checkpoint_root` for private checkpoint archives, and `runtime_home`: a private directory, which the helper creates with mode `0700` when it is missing. microsandbox places Unix sockets under it, so keep its path within 48 bytes; the installer refuses a longer one for its own nodes. + Before the first microsandbox registration, initialize the empty `checkpoint_root` as the node service account using the same release's source checkout: `PYTHONPATH=deploy/node python3 -c 'from node_install import prepare_checkpoint_root; prepare_checkpoint_root("/absolute/private/checkpoints")'`. This reuses the installer's exclusive UUID-marker initialization and ownership checks. Initialize a shared store once; other nodes use the existing marker. Never replace the marker or initialize over restored or nonempty unowned storage. 5. Register, then run the node under the host's service supervisor, with real absolute paths: ```sh @@ -148,7 +151,7 @@ The node connects out to Core; Core needs no SSH or Docker TCP access to the hos ## When a node host fails -A restarted node service keeps its identity and finds its existing sandboxes again. Core never replaces a missing sandbox by itself, and never moves a Session to another node: the Session's resources show as **Node disconnected** or **Sandbox resource missing** until the original host and its storage are back, or you archive the Session. A lost node state directory is a recovery incident: restore it from its [backup](./operations.md#back-up) together with the database and the provider storage, rather than registering the host again over existing resources. +A restarted node service keeps its identity and finds its existing sandboxes again. Core does not take over an active or unconfirmed writer merely because its node is unreachable or its sandbox is missing: those resources show as **Node disconnected** or **Sandbox resource missing** while ownership remains unresolved. A settled retained checkpoint can follow the [checkpoint transfer contract](../sandbox-provider.md#checkpoint-transfer), and confirmed release can permit [cold replacement](../sandbox-provider.md#cold-replacement-after-checkpoint-retention). A lost node state directory is a recovery incident: restore it from its [backup](./operations.md#back-up) together with the database and the provider storage, rather than registering the host again over existing resources. ## Troubleshooting diff --git a/docs/getting-started/operations.md b/docs/getting-started/operations.md index 33cadfe6a..c3e5b8ca7 100644 --- a/docs/getting-started/operations.md +++ b/docs/getting-started/operations.md @@ -130,6 +130,8 @@ A recoverable backup contains one consistent, stopped-write set. A database dump - Every independent workspace namespace selected by current or retained objects, even when it is outside the installation volume. Back up its entire root, including `.oac-storage-root`, object identities, staging, live data, trash and deletion markers. Preserve numeric ownership, modes, links and extended attributes, including every `user.*` attribute. Copying only `live/data` loses lifecycle evidence and can resurrect deleted identities. [Workspace storage](../workspace-provider.md#kernel-nfs-adapter) owns the layout and ownership requirements. - Each node's state directory, `/var/lib/oac-node/.oac/nodes//`, and its provider storage: Docker volumes or microsandbox's store. Include all compute and snapshot resources still referenced by the database. See [when a node host fails](./nodes.md#when-a-node-host-fails). +- Each private `checkpoint_root` in its entirety, including `.oac-checkpoint-store`, archives, operation journals, lock files and tombstones, together with the matching Core database. It is separate from the native SDK store and independent workspace namespace. Preserve its identity and private ownership; the [checkpoint transfer contract](../sandbox-provider.md#checkpoint-transfer) owns its lifecycle. + ### Establish a stopped-write window 1. Block new application input, live file access and administrative mutations at the installation's ingress. Let active execution, initialization, file operations and native cleanup finish. Stop other applications or host processes that can write to the same filesystem. diff --git a/docs/sandbox-provider.md b/docs/sandbox-provider.md index a821f3343..d48ae3eeb 100644 --- a/docs/sandbox-provider.md +++ b/docs/sandbox-provider.md @@ -180,7 +180,7 @@ Before releasing the execution lease, the coordinator stops accepting work and c Placement is automatic: Session creation commits durable pending work, and the common scheduler reserves a compatible node before allocation. The [deployment contract](../contracts/agents-api/sandbox-deployment.md#generation-ownership-and-rollout) owns waiting, ordering and generation selection. Callers cannot choose a node, and an active allocation keeps its original node even while it is offline. After confirmed allocation release, eligible retained Sessions can reserve compatible capacity again, including on another node. Node capacity counts pending reservations and unresolved resources, and new placement and a suspended-to-restoring transition share a database lock. Unknown operations keep their reservations, source teardown must be confirmed before active capacity is released, and confirmed cleanup releases placement capacity. Retained ownership needs exact provider evidence: a socket path, a missing instance or an empty listing never proves cleanup or authorizes a replacement. -The deployment's CPU, memory and disk settings, `max_active`, `max_retained` and the snapshot retention bound each node. Cold replacement after confirmed release does not provide node-level drain, cross-node memory restore, failover from an unreachable writer, concurrent writers, multi-active Core, autoscaling or snapshot replication. Node removal is refused while the node holds allocations, snapshots, reservations, unknown results or cleanup, and offline ownership is kept. +The deployment's CPU, memory and disk settings, `max_active`, `max_retained` and the snapshot retention bound each node. Cold replacement after confirmed release does not provide node-level drain, failover from an unreachable writer, concurrent writers, multi-active Core, autoscaling or snapshot replication. Node removal is refused while the node holds allocations, snapshots, reservations, unknown results or cleanup, and offline ownership is kept. ### Suspension @@ -190,7 +190,17 @@ Capacity pressure can suspend an already idle node allocation before the normal The Worker lease, the Session lock and the per-node gates own suspension for every provider. New Turn claims, file-write intents and capture admission serialize under the Session lock and share one compute-phase check; new pending work cancels a capture and wakes the same source. Normal preparation waits for the compute phase to be running, after the authenticated resume handshake, and pending input stays pending when its promotion conflicts with a lifecycle transition. Compute phases and revision-checked receipts live on the allocation. Core persists quiesce, capture and restore intent before the effect, only a fresh receipt performs a capture or restore, and recovery observes the exact attempt without retrying an unknown creation, capture or restore. A consumed snapshot never rolls a running generation back. Deletion, revocation and retention expiry win over wake, up to the final database compare-and-swap, and unknown cleanup identities are kept until owned resources are confirmed absent. Consumed artifacts and old compute are deleted, so suspension cycles never build a chain of writable disks. -Queued work and live Environment file access wake a suspended Environment; history and published Artifact reads do not. Live file requests wait for the current Runtime’s credential-authorized connection and Harness declaration before entering the file work queue; a completed compute Create alone is not Runtime readiness. Planned suspension uses an Environment and suspension token on the daemon connection. A PID and start-time fenced local control signal (`RunCommandCompute`) wakes the parked daemon, which authenticates again before admitting work. A transient disconnect before confirmation retries the same armed suspension with bounded attempts and backoff; a permanent authentication or protocol rejection closes it. The retained checkpoint resumes on its original node. Core owns the snapshot's retention deadline, and the daemon has no timer for it. A lost quiesce acknowledgement may thaw the same source through explicit rollback but never authorizes capturing it. +Queued work and live Environment file access wake a suspended Environment; history and published Artifact reads do not. Live file requests wait for the current Runtime’s credential-authorized connection and Harness declaration before entering the file work queue; a completed compute Create alone is not Runtime readiness. Planned suspension uses an Environment and suspension token on the daemon connection. A PID and start-time fenced local control signal (`RunCommandCompute`) wakes the parked daemon, which authenticates again before admitting work. A transient disconnect before confirmation retries the same armed suspension with bounded attempts and backoff; a permanent authentication or protocol rejection closes it. A retained checkpoint resumes on its original node when it is eligible, or on another eligible node through the checkpoint transfer rules below. Core owns the snapshot's retention deadline, and the daemon has no timer for it. A lost quiesce acknowledgement may thaw the same source through explicit rollback but never authorizes capturing it. + +### Checkpoint transfer + +A checkpoint-capable generation reports `CheckpointCompatibility`: opaque `ArtifactDomain` and `ExecutionClass` tokens. Every verified `SnapshotIdentity` carries the same qualification. Core compares these values without interpreting CPU features, filesystem paths or storage implementations. A restore destination must be online and ready for the allocation's immutable deployment generation, match both tokens, and have active capacity plus a retained slot when moving from another node. The allocation, Device and Session identities remain unchanged. The Session and deployment transaction commits the destination route, placement and exact restore intent before target-side native work. No capacity or compatible destination leaves the retained snapshot owned until its configured deadline. The target must already hold that exact generation; Core does not automatically prepare historical generations on new nodes. During upgrades, retain eligible nodes and their generation providers until their checkpoint retention obligations end. + +`Suspend` may report `SourceStopped` only after publishing a verified full archive durably, stopping the exact source compute and settling source-local capture and cleanup obligations. Core persists the snapshot before marking it suspended; native absence alone is insufficient. `Suspend.ObserveOnly` may finish archive publication and source cleanup for an already verified snapshot of the same operation, but cannot capture another snapshot or stop a source that has resumed running. If a previously published archive is missing or damaged, `ObserveOnly` may return its exact durable ownership receipt with `Status: unknown` and `SourceStopped: false` for cleanup; that receipt proves neither current archive usability nor source absence. `Resume` still verifies the archive before execution. After the source-settlement barrier, source-node unavailability does not prevent restoration. Deletion and ordinary expiry can transfer an idle suspended allocation's cleanup route to an online node in the same artifact domain; deletion does not require execution-class compatibility. An unknown running or restoring writer never moves merely because its node is unreachable. + +Recovery uses `Resume.ObserveOnly` for the persisted target and operation. `RestoreAttemptClosed` is an exact operation ID, not a generic absence flag: the adapter may return it only after durably closing an attempt that never entered native execution and fencing every late request for that ID. Core then persists a new operation ID before attempting execution. An admitted attempt whose outcome is unknown retains the same target and ownership; a missing native listing, helper death or timeout does not authorize replay. User deletion still requires exact native settlement before resources and ownership can be released. + +The microsandbox adapter requires an explicit private `checkpoint_root`, outside every guest-accessible filesystem. Its directory is mode `0700`; a private namespace marker identifies the actual archive store. The default installation creates a node-local private store. Cross-node restoration requires operators to mount the same private store on each participating node; mount paths may differ. Runtime state directories and workspace bindings are never used to infer this root. The adapter verifies the archive, external filesystem identity and native execution class again before restoration. Restored RAM and file descriptors do not promise continuity of external TCP connections or replay safety for application side effects. ### Cold replacement after checkpoint retention @@ -252,4 +262,4 @@ The node uses the explicit Unix socket in its [provider configuration](./configu `DeploymentPolicy.Workspace` declares supported attachment requirements; absence means external storage is unsupported. The derived immutable `DeploymentSpec.Workspace` receipt selects the generation mode and capabilities; its [deployment workflow](../contracts/agents-api/sandbox-deployment.md#resources) never copies filesystem configuration into the generation. Microsandbox requires `host_directory` and `user_xattr`; Docker and E2B reject external attachments. `ValidateWorkspacePolicy` uses the filesystem protocol's combination validator: an external filesystem without an enforced quota rejects a positive `EnvironmentDiskMiB`, while zero requests no quota. Root disk capacity is still required. Without an external declaration, the existing owned disk bounds apply. -Microsandbox resolves each Create and Resume binding, including observation of an interrupted restore, and passes only the local directory and immutable ObjectID to its private helper. The helper binds the whole `/environment`; its retained contents and private native history follow the [workspace filesystem scope](workspace-provider.md#lifetime-and-filesystem-scope) and [Runtime resource directories](configuration.md#runtime-resource-directories). Private HOME, credentials and the compute root are not a durable history store. Explicit mode, object and path labels qualify native resources; mount shape alone never selects a mode. Same-node checkpoint restore verifies the snapshot's object, remaps `/environment` through SDK `Volumes`, requires strict external mount policy and rejects every restore warning. Partial restore errors retain the exact target identity for cleanup and never establish readiness. Unknown outcomes remain observations of the original operation. Native Bind checkpoint and guest-local flock continuity evidence does not establish cross-VM fencing or cross-node memory restore. +Microsandbox resolves each Create and Resume binding, including observation of an interrupted restore, and passes only the local directory and immutable ObjectID to its private helper. The helper binds the whole `/environment`; its retained contents and private native history follow the [workspace filesystem scope](workspace-provider.md#lifetime-and-filesystem-scope) and [Runtime resource directories](configuration.md#runtime-resource-directories). Private HOME, credentials and the compute root are not a durable history store. Explicit mode, object and path labels qualify native resources; mount shape alone never selects a mode. Checkpoint restore verifies the snapshot's object, remaps `/environment` through SDK `Volumes`, requires strict external mount policy and rejects every restore warning. Partial restore errors retain the exact target identity for cleanup and never establish readiness. Unknown outcomes remain observations of the original operation. Native Bind checkpoint and guest-local flock continuity do not establish fencing against an unknown writer. diff --git a/docs/zh/getting-started/nodes.md b/docs/zh/getting-started/nodes.md index d7c572de2..bf4135da6 100644 --- a/docs/zh/getting-started/nodes.md +++ b/docs/zh/getting-started/nodes.md @@ -1,7 +1,7 @@ --- title: "添加和管理节点" source: docs/getting-started/nodes.md -source_hash: 9b75587ccab0088003e8bd12afe477af0e72fe6650045f5283700248f9df327e +source_hash: a31edf7908ecd0a848b562bbceda3fd872436473d5ec1f2bedfba8719ac9e2a1 --- 节点是一台 Linux 主机,在沙箱后端为 Docker 或 microsandbox 时,为 Core 托管 Session 运行沙箱。Core 将新 Session 分配给有空余容量的节点;节点创建沙箱,沙箱回连 Core。E2B 不需要节点。应用为自己的 Session 连接的机器是[自托管执行器](self-hosted.md),而不是节点。 @@ -16,6 +16,8 @@ source_hash: 9b75587ccab0088003e8bd12afe477af0e72fe6650045f5283700248f9df327e Core 主机与其他主机一样加入:要在它上面运行沙箱,将它添加为节点。 +microsandbox 跨节点检查点恢复要求在每个参与节点的安装命令加入 `--checkpoint-root /absolute/private/shared-store`。安装前将同一个私有 store 挂载到这些路径,并只允许 node 服务账户访问。默认使用节点本地私有 store,只支持在该节点恢复。检查点 store 独立于 workspace 存储,绝不能允许 guest 访问;[检查点转移契约](../sandbox-provider.md#checkpoint-transfer) 定义兼容性和保留要求。 + 对于使用独立工作区存储的 microsandbox,应在注册前完成此主机上的 [NFS 挂载和服务账户配置](../configuration.md#independent-workspace-storage)。节点随每次 binding 接收所选不可变文件系统配置;不要单独编写节点存储设置。 ## 添加节点 {#add-a-node} @@ -129,7 +131,8 @@ root 只准备账号、组和服务单元;其他操作(包括 Docker 网络 3. 使用令牌读取节点配置,不会消耗令牌:`GET /api/v1/sandbox-node/configuration`,带 `Authorization: Bearer `。 4. 写入私有提供商文件。从响应复制 `provider`、`installation_id`、`core_url`、`generation` 和 `specification`,并添加 `native` 对象,写入该提供商的主机设置。适配器从 `specification` 读取沙箱规格、Runtime 镜像和产物哈希: - Docker:[Docker 节点配置](../configuration.md#docker-node-configuration)中的字段,其中 `host` 是显式 Unix 套接字,`image` 是导入的 Runtime 镜像的本地 ID,`seccomp_file` 是绝对路径。 - - microsandbox:绝对路径 `helper_path`、`runtime_path` 和 `firmware_path`;`network` 策略;以及 `runtime_home` 私有目录。目录缺失时辅助程序以 `0700` 创建。microsandbox 在其中放置 Unix 套接字,因此路径不要超过 48 字节;安装程序对自管节点拒绝更长路径。 + - microsandbox:绝对路径 `helper_path`、`runtime_path` 和 `firmware_path`;`network` 策略;私有检查点归档的显式 `checkpoint_root`;以及 `runtime_home` 私有目录。目录缺失时辅助程序以 `0700` 创建。microsandbox 在其中放置 Unix 套接字,因此路径不要超过 48 字节;安装程序对自管节点拒绝更长路径。 + 首次注册 microsandbox 前,以 node 服务账户在相同发行版本的源码 checkout 中初始化空的 `checkpoint_root`:`PYTHONPATH=deploy/node python3 -c 'from node_install import prepare_checkpoint_root; prepare_checkpoint_root("/absolute/private/checkpoints")'`。它复用安装程序的 UUID marker 排他初始化与所有权检查。共享 store 只初始化一次,其他 node 使用已有 marker。不得替换 marker,也不得在恢复后的存储或非空且无所有权证明的目录上重新初始化。 5. 使用真实绝对路径注册,然后通过主机服务管理器运行节点: ```sh @@ -150,7 +153,7 @@ root 只准备账号、组和服务单元;其他操作(包括 Docker 网络 ## 节点主机故障时 {#when-a-node-host-fails} -重启节点服务会保留身份并重新发现已有沙箱。Core 不会自行替换缺失沙箱,也不会将 Session 移到其他节点:Session 资源显示 **Node disconnected** 或 **Sandbox resource missing**,直到原主机及存储恢复,或你归档 Session。丢失节点状态目录属于恢复事件:从[备份](operations.md#back-up)恢复,并同时恢复数据库和提供商存储;不要在已有资源上重新注册主机。 +重启节点服务会保留身份并重新发现已有沙箱。Core 不会仅因节点不可达或沙箱缺失就接管活动或未确认的 writer:所有权未结清时,这些资源显示 **Node disconnected** 或 **Sandbox resource missing**。已结清的保留检查点可以遵循[检查点转移契约](../sandbox-provider.md#checkpoint-transfer),确认释放后可以按条件进行[冷替换](../sandbox-provider.md#cold-replacement-after-checkpoint-retention)。丢失节点状态目录属于恢复事件:从[备份](operations.md#back-up)恢复,并同时恢复数据库和提供商存储;不要在已有资源上重新注册主机。 ## 问题排查 {#troubleshooting} diff --git a/docs/zh/getting-started/operations.md b/docs/zh/getting-started/operations.md index 8ab281776..4316ac34d 100644 --- a/docs/zh/getting-started/operations.md +++ b/docs/zh/getting-started/operations.md @@ -1,7 +1,7 @@ --- title: "运维" source: docs/getting-started/operations.md -source_hash: acd6e73b0bdf43f3bb5adfc375c8938d1b8169a326620076242c844ab8b4766b +source_hash: 591196d61ed9946435d07afcbbcc2f53d64b9db0cf6b4806ddfb0f85a16974ba --- 安装运维人员负责 Core 主机、存储和可用性。节点主机运行各自的服务;参阅[节点](nodes.md)。设置见[配置参考](../configuration.md)。 @@ -132,6 +132,8 @@ Core 记录每次公开资源写入所使用的密钥;历史保留策略为 [` - 当前或保留对象所使用的每个独立工作区命名空间,即使其位于安装卷之外。备份整个根目录,包括 `.oac-storage-root`、对象标识、staging、live 数据、trash 和删除标记。保留数值所有者、权限、链接及扩展属性,包括全部 `user.*` 属性。仅复制 `live/data` 会丢失生命周期证据,并可能使已删除标识复活。[工作区存储](../workspace-provider.md#kernel-nfs-adapter)负责定义布局与所有权要求。 - 各节点状态目录 `/var/lib/oac-node/.oac/nodes//` 及其 Provider 存储:Docker 卷或 microsandbox 存储。包含数据库仍然引用的全部计算资源和快照。参见[节点主机故障时](./nodes.md#when-a-node-host-fails)。 +- 每个私有 `checkpoint_root` 的全部内容,包括 `.oac-checkpoint-store`、归档、operation journal、lock 文件与 tombstone,并与匹配的 Core 数据库一起保留。它独立于原生 SDK store 和工作区 namespace。保留其身份和私有所有权;[检查点转移契约](../sandbox-provider.md#checkpoint-transfer)定义其生命周期。 + ### 建立停止写入窗口 {#establish-a-stopped-write-window} 1. 在安装入口阻止新应用 input、实时文件访问和管理变更。等待活动执行、初始化、文件操作和原生清理完成。停止其他能够写入同一文件系统的应用或主机进程。 diff --git a/docs/zh/sandbox-provider.md b/docs/zh/sandbox-provider.md index 48562ef57..fc4a3eba2 100644 --- a/docs/zh/sandbox-provider.md +++ b/docs/zh/sandbox-provider.md @@ -1,7 +1,7 @@ --- title: "添加 Sandbox Provider" source: docs/sandbox-provider.md -source_hash: 567cd130134a45f483b156d7f2b74d246b191dd5db77b976976c0d710414dd37 +source_hash: 08e51f80e7b4a945bceacd98135c52f5f97d5ecb85bfc0b027b3fe9933ce7ce0 --- **Sandbox Provider** 为 Core 管理的 Environment 提供 Runtime daemon 运行所需的外层计算资源,以及启动 daemon 的有界引导流程。本指南说明如何添加 Provider,并作为 Core 驱动 Provider 的参考。接口为 [`SandboxProvider`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/services/core/internal/sandbox/sandbox_provider.go)。 @@ -182,7 +182,7 @@ allocation scan 在应用 32 行分页限制前按 node 过滤,pending scan jo placement 自动完成:Session 创建提交持久的 pending 工作,共同调度器在 allocation 之前预留兼容节点。[部署契约](../../contracts/agents-api/zh/sandbox-deployment.md#generation-ownership-and-rollout) 定义等待、排序与代次选择。调用方不能选择 node;活动 allocation 即使在 node 离线时也保留原 node。确认 allocation 释放后,符合条件的保留 Session 可以重新预约兼容容量,包括其他 node。Node capacity 计入 pending reservation 和未决资源,新 placement 与 suspended-to-restoring 转移共享数据库锁。未知操作保留预约,source teardown 必须确认后才能释放 active capacity,确认 cleanup 后释放 placement capacity。保留所有权需要精确 provider evidence:socket path、缺失 instance 或空列表都不证明 cleanup,也不授权 replacement。 -部署的 CPU、memory、disk 设置、`max_active`、`max_retained` 和 snapshot retention 限制每个 node。确认释放后的冷替换不提供 node-level drain、跨 node 内存恢复、不可达写入方的故障转移、并发写入方、multi-active Core、autoscaling 或 snapshot replication。node 持有 allocation、snapshot、reservation、unknown result 或 cleanup 时拒绝移除 node,离线 ownership 保留。 +部署的 CPU、memory、disk 设置、`max_active`、`max_retained` 和 snapshot retention 限制每个 node。确认释放后的冷替换不提供 node-level drain、不可达写入方的故障转移、并发写入方、multi-active Core、autoscaling 或 snapshot replication。node 持有 allocation、snapshot、reservation、unknown result 或 cleanup 时拒绝移除 node,离线 ownership 保留。 ### 暂停 {#suspension} @@ -192,7 +192,17 @@ Core 用同一个共享 policy 暂停每个声明 checkpoint 支持的 provider Worker lease、Session lock 与 per-node gate 对每个 provider 负责 suspension。新 Turn claim、file-write intent 和 capture admission 在 Session lock 下串行化,共享一个 compute-phase 检查;新 pending work 取消 capture 并唤醒同一 source。正常 preparation 在经过认证的 resume handshake 后等待 compute phase 为 running;pending input 的 promotion 与 lifecycle transition 冲突时保持 pending。compute phase 和 revision-checked receipt 位于 allocation。Core 在 effect 前持久化 quiesce、capture 和 restore intent,仅新 receipt 执行 capture 或 restore,恢复观察精确 attempt,不重试未知 creation、capture 或 restore。已消费 snapshot 不让 running generation 回滚。删除、撤销和 retention expiry 优先于 wake,一直持续到最终数据库 compare-and-swap;未知 cleanup identity 保留,直到确认所属资源不存在。已消费 artifact 和旧 compute 被删除,因此暂停循环不累积可写磁盘链。 -排队工作和实时 Environment file access 唤醒 suspended Environment;history 和已发布 Artifact read 不唤醒。实时文件请求在进入文件工作队列前,等待当前 Runtime 通过 credential 授权的连接及 Harness 声明;compute Create 完成并不代表 Runtime 已就绪。计划暂停在 daemon 连接上使用 Environment 和 suspension token。受 PID 与 start-time fencing 的本地 control signal(`RunCommandCompute`)唤醒 parked daemon,daemon 在准入工作前重新认证。确认前临时断连通过有界 attempt 和 backoff 重试同一已 armed suspension;永久认证或协议拒绝则关闭。保留的检查点在原 node 恢复。Core 负责 snapshot retention deadline,daemon 没有相应 timer。quiesce 确认丢失时可以通过明确 rollback 解冻同一 source,但不授权 capture。 +排队工作和实时 Environment file access 唤醒 suspended Environment;history 和已发布 Artifact read 不唤醒。实时文件请求在进入文件工作队列前,等待当前 Runtime 通过 credential 授权的连接及 Harness 声明;compute Create 完成并不代表 Runtime 已就绪。计划暂停在 daemon 连接上使用 Environment 和 suspension token。受 PID 与 start-time fencing 的本地 control signal(`RunCommandCompute`)唤醒 parked daemon,daemon 在准入工作前重新认证。确认前临时断连通过有界 attempt 和 backoff 重试同一已 armed suspension;永久认证或协议拒绝则关闭。保留的检查点在原 node 仍具资格时优先恢复,也可以按以下检查点转移规则在其他符合条件的 node 恢复。Core 负责 snapshot retention deadline,daemon 没有相应 timer。quiesce 确认丢失时可以通过明确 rollback 解冻同一 source,但不授权 capture。 + +### 检查点转移 {#checkpoint-transfer} + +支持检查点的 generation 报告 `CheckpointCompatibility`,包含不透明的 `ArtifactDomain` 与 `ExecutionClass` token。每个已验证 `SnapshotIdentity` 携带同样的资格。Core 只比较这些值,不解释 CPU 特性、文件系统路径或存储实现。恢复目标必须在线、对 allocation 的不可变 deployment generation 已就绪、两个 token 均匹配,并有 active 容量;跨 node 时还需 retained slot。allocation、Device 和 Session 身份保持不变。Session 与 deployment 事务先提交目标路由、placement 和精确 restore intent,再执行目标侧原生操作。没有容量或兼容目标时,保留快照的所有权一直持续到配置期限。目标必须已持有该精确 generation;Core 不会在新节点自动准备历史 generation。升级期间应保留符合条件的节点及其 generation provider,直到检查点保留义务结束。 + +`Suspend` 只有在持久发布已验证的完整归档、停止精确 source compute,并结清 source 本地 capture 与 cleanup 义务后,才能报告 `SourceStopped`。Core 先持久化快照,再标记 suspended;仅原生资源不存在并不足够。`Suspend.ObserveOnly` 可以为同一 operation 已验证的快照完成归档发布和源清理,但不能重新捕获快照,也不能停止已经恢复 running 的源。此前已发布的归档缺失或损坏时,`ObserveOnly` 可以返回其精确持久 ownership receipt,标记 `Status: unknown` 和 `SourceStopped: false`,供清理使用;该 receipt 不证明归档当前可用,也不证明源不存在。`Resume` 仍在执行前验证归档。完成源结清屏障后,source node 不可用不再阻止恢复。删除和正常到期可以将空闲 suspended allocation 的清理路由转给同一 artifact domain 内的在线 node;删除不要求 execution class 匹配。未知 running 或 restoring writer 不会仅因 node 不可达而迁移。 + +恢复用 `Resume.ObserveOnly` 观察已持久化的 target 与 operation。`RestoreAttemptClosed` 是精确 operation ID,不是通用 absence 标志:adapter 只有在持久关闭从未进入原生执行的 attempt,并隔离该 ID 的所有迟到请求后才能返回。Core 随后先持久化新的 operation ID,再尝试执行。已 admitted 但结果未知的 attempt 保留相同 target 与所有权;原生列表缺失、helper 死亡或超时都不授权重放。用户删除仍需精确原生 settlement,之后才能释放资源与所有权。 + +microsandbox adapter 要求显式的私有 `checkpoint_root`,位于所有 guest 可访问文件系统之外。目录权限为 `0700`,私有 namespace marker 标识真实归档存储。默认安装创建 node 本地私有 store。跨 node 恢复要求运维为参与 node 挂载同一个私有 store,挂载路径可以不同。Runtime state 目录与 workspace binding 都不用于推导该 root。adapter 在恢复前再次验证归档、外部文件系统身份和原生 execution class。恢复 RAM 与文件描述符不保证外部 TCP 连接连续,也不保证应用副作用可安全重放。 ### 检查点保留期后的冷替换 {#cold-replacement-after-checkpoint-retention} @@ -254,4 +264,4 @@ node 使用 [provider 配置](configuration.md#docker-node-configuration)中的 `DeploymentPolicy.Workspace` 声明挂载要求;未声明表示不支持外部存储。派生的不可变 `DeploymentSpec.Workspace` 回执选择代次的模式和能力;[部署流程](../../contracts/agents-api/zh/sandbox-deployment.md#resources)不会将文件系统配置复制进代次。Microsandbox 要求 `host_directory` 和 `user_xattr`;Docker 和 E2B 拒绝外部挂载。`ValidateWorkspacePolicy` 使用文件系统协议的组合校验器:不强制容量配额的外部文件系统拒绝正值 `EnvironmentDiskMiB`,零表示不请求配额。根磁盘容量仍为必需。未提供外部声明时,继续使用自有磁盘的容量边界。 -Microsandbox 在每次 Create 和 Resume(包括观察中断的恢复)时解析绑定,仅向私有 helper 传递本地目录和不可变 ObjectID。Helper 挂载整个 `/environment`;保留内容与私有原生历史遵循[工作区文件系统范围](workspace-provider.md#lifetime-and-filesystem-scope)和 [Runtime 资源目录](configuration.md#runtime-resource-directories)。私有 HOME、凭据和计算根磁盘不是持久历史存储。原生资源通过明确的模式、对象和路径标签校验,不根据挂载形状推断模式。同 node 检查点恢复校验快照对象,通过 SDK `Volumes` 重映射 `/environment`,要求严格外部挂载策略,并拒绝任何恢复警告。恢复部分失败时保留精确目标身份用于清理,不证明就绪。未知结果继续观察原操作。原生 Bind 检查点及 guest 内 flock 连续性证据不证明跨 VM 隔离或跨 node 内存恢复。 +Microsandbox 在每次 Create 和 Resume(包括观察中断的恢复)时解析绑定,仅向私有 helper 传递本地目录和不可变 ObjectID。Helper 挂载整个 `/environment`;保留内容与私有原生历史遵循[工作区文件系统范围](workspace-provider.md#lifetime-and-filesystem-scope)和 [Runtime 资源目录](configuration.md#runtime-resource-directories)。私有 HOME、凭据和计算根磁盘不是持久历史存储。原生资源通过明确的模式、对象和路径标签校验,不根据挂载形状推断模式。检查点恢复校验快照对象,通过 SDK `Volumes` 重映射 `/environment`,要求严格外部挂载策略,并拒绝任何恢复警告。恢复部分失败时保留精确目标身份用于清理,不证明就绪。未知结果继续观察原操作。原生 Bind 检查点及 guest 内 flock 连续性不证明对未知 writer 的隔离。 diff --git a/services/core/cmd/sandbox-node/main.go b/services/core/cmd/sandbox-node/main.go index 2a6d3b032..4e955be99 100644 --- a/services/core/cmd/sandbox-node/main.go +++ b/services/core/cmd/sandbox-node/main.go @@ -93,8 +93,12 @@ func run(ctx context.Context, args []string) error { } expected := node.Identity{SpecificationDigest: built.SpecificationDigest, DeploymentGeneration: config.Generation, InstallationID: built.InstallationID, Provider: config.Provider, BackendFingerprint: built.BackendFingerprint} probe := func(ctx context.Context) (node.Health, error) { - err := built.Probe(ctx) - return node.Health{ProviderReady: err == nil}, err + checkpoint, err := built.Probe(ctx) + status := sandbox.GenerationStatus{Generation: config.Generation, SpecificationDigest: built.SpecificationDigest, State: "ready", Checkpoint: checkpoint} + if err != nil { + status.State, status.Diagnostic, status.Checkpoint = "failed", sandbox.NodeDiagnostic(err), nil + } + return node.Health{ProviderReady: err == nil, Generations: []sandbox.GenerationStatus{status}}, err } if args[0] == "register" { if !filepath.IsAbs(*tokenFile) { diff --git a/services/core/cmd/server/managed_nodes.go b/services/core/cmd/server/managed_nodes.go index b908bbab6..c514c7e3d 100644 --- a/services/core/cmd/server/managed_nodes.go +++ b/services/core/cmd/server/managed_nodes.go @@ -67,7 +67,7 @@ func configureManagedNodes(nodes *deployment.Service, reader deployment.Reader, if err := owner(ctx); err != nil { return err } - return nodes.Heartbeat(ctx, n.NodeID, connection, epoch, nodeHealthRecord(health)) + return nodes.Heartbeat(ctx, n.NodeID, connection, epoch, nodeHealthRecord(health), health.Generations) }, }) result.setup = &managedSetup{processPaths: config.ProviderPaths, registry: registry, deployment: nodes, allocations: reader, hub: result.hub, installationID: config.InstallationID, runtimeAPI: config.PublicOrigin.RuntimeAPI()} diff --git a/services/core/internal/db/queries/node_generations.sql b/services/core/internal/db/queries/node_generations.sql index 7b917e529..436268027 100644 --- a/services/core/internal/db/queries/node_generations.sql +++ b/services/core/internal/db/queries/node_generations.sql @@ -10,10 +10,10 @@ UNION ALL SELECT g.provider_kind,g.specification FROM runtime_deployment_generat LIMIT 1; -- name: UpsertNodeGenerationStatus :exec -INSERT INTO runtime_node_generation_status(node_id,generation,specification_digest,connection_id,owner_epoch,state,diagnostic) -VALUES($1,$2,$3,$4,$5,$6,$7) +INSERT INTO runtime_node_generation_status(node_id,generation,specification_digest,connection_id,owner_epoch,state,diagnostic,checkpoint) +VALUES($1,$2,$3,$4,$5,$6,$7,$8) ON CONFLICT(node_id,generation) DO UPDATE SET specification_digest=EXCLUDED.specification_digest,connection_id=EXCLUDED.connection_id, - owner_epoch=EXCLUDED.owner_epoch,state=EXCLUDED.state,diagnostic=EXCLUDED.diagnostic,observed_at=clock_timestamp(); + owner_epoch=EXCLUDED.owner_epoch,state=EXCLUDED.state,diagnostic=EXCLUDED.diagnostic,checkpoint=EXCLUDED.checkpoint,observed_at=clock_timestamp(); -- name: PromoteNodeServingGeneration :exec UPDATE runtime_nodes SET ready_generation=$2 WHERE id=$1 AND removed_at IS NULL @@ -33,3 +33,13 @@ SELECT EXISTS(SELECT 1 FROM runtime_node_generation_status g JOIN runtime_nodes -- name: DeleteNodeGenerationStatus :exec DELETE FROM runtime_node_generation_status WHERE node_id=$1 AND generation=$2; + +-- name: ListCheckpointGenerationNodes :many +SELECT n.id, g.checkpoint +FROM runtime_nodes n +JOIN runtime_node_generation_status g ON g.node_id=n.id +CROSS JOIN runtime_deployment d +WHERE g.generation=$1 AND n.installation_id=d.installation_id AND n.removed_at IS NULL +AND n.connection_id=g.connection_id AND n.connected_epoch=g.owner_epoch AND g.owner_epoch=d.owner_epoch +AND n.last_seen_at>clock_timestamp()-interval '45 seconds' AND g.state='ready' +ORDER BY n.id; diff --git a/services/core/internal/db/queries/runtime_suspension.sql b/services/core/internal/db/queries/runtime_suspension.sql index 5bb294a8d..1bd60f42d 100644 --- a/services/core/internal/db/queries/runtime_suspension.sql +++ b/services/core/internal/db/queries/runtime_suspension.sql @@ -53,3 +53,24 @@ SELECT EXISTS ( SELECT 1 FROM runtime_allocations a JOIN environments e ON e.id = a.environment_id WHERE e.session_id = $1 AND a.state <> 'released' AND a.node_id IS NOT NULL )::boolean; + +-- name: MoveSuspendedRuntimeCompute :one +WITH moved AS ( + UPDATE runtime_placements p SET node_id=sqlc.arg(destination)::uuid + FROM runtime_allocations a + WHERE a.id=sqlc.arg(id) AND a.environment_id=p.environment_id + AND p.node_id=sqlc.arg(source)::uuid AND p.released_at IS NULL + AND p.deployment_generation=a.deployment_generation + AND a.node_id=p.node_id AND a.compute_revision=sqlc.arg(revision) + AND a.compute_phase='suspended' + AND ((a.state='running' AND sqlc.arg(phase)::text='restoring' + AND a.compute_retained_until>clock_timestamp() + AND EXISTS(SELECT 1 FROM environments e WHERE e.id=a.environment_id AND e.initialization='complete')) + OR (a.state='cleanup_pending' AND sqlc.arg(phase)::text='suspended')) + RETURNING a.id +) +UPDATE runtime_allocations a +SET node_id=sqlc.arg(destination)::uuid, compute_phase=sqlc.arg(phase), + compute_state=sqlc.arg(state)::jsonb, compute_revision=a.compute_revision+1, + compute_phase_changed_at=CASE WHEN a.compute_phase=sqlc.arg(phase)::text THEN a.compute_phase_changed_at ELSE clock_timestamp() END +FROM moved WHERE a.id=moved.id RETURNING a.*; diff --git a/services/core/internal/db/queries/runtime_suspension_pressure.sql b/services/core/internal/db/queries/runtime_suspension_pressure.sql index e46b2fd1f..580c00813 100644 --- a/services/core/internal/db/queries/runtime_suspension_pressure.sql +++ b/services/core/internal/db/queries/runtime_suspension_pressure.sql @@ -5,15 +5,10 @@ WHERE a.node_id IS NOT NULL AND a.state <> 'released' AND (a.compute_phase IN ('quiescing', 'suspending') OR (a.compute_retained_until <= clock_timestamp() AND a.compute_phase <> 'disabled')); --- name: ListWaitingRuntimeRestoreGenerations :many -SELECT DISTINCT a.deployment_generation -FROM runtime_allocations a -JOIN environments e ON e.id = a.environment_id -JOIN sessions s ON s.id = e.session_id -WHERE a.node_id = $1 AND a.state = 'running' AND a.compute_phase = 'suspended' - AND a.compute_retained_until > clock_timestamp() - AND s.deleted_at IS NULL AND e.status NOT IN ('failed', 'expired') - AND (a.compute_wake_requested OR EXISTS ( - SELECT 1 FROM environment_input_reservations r - WHERE r.session_id = s.id AND r.state = 'pending' AND r.deadline > clock_timestamp() - )); +-- name: ListWaitingCheckpointRestores :many +SELECT DISTINCT a.node_id, a.deployment_generation, (a.compute_state->'snapshot'->'Compatibility')::jsonb AS checkpoint +FROM runtime_allocations a JOIN environments e ON e.id=a.environment_id JOIN sessions s ON s.id=e.session_id +WHERE a.node_id IS NOT NULL AND a.state='running' AND a.compute_phase='suspended' + AND a.compute_retained_until>clock_timestamp() AND s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND a.compute_state->'snapshot'->'Compatibility' IS NOT NULL + AND (a.compute_wake_requested OR EXISTS(SELECT 1 FROM environment_input_reservations r WHERE r.session_id=s.id AND r.state='pending' AND r.deadline>clock_timestamp())); diff --git a/services/core/internal/db/sqlc/models.go b/services/core/internal/db/sqlc/models.go index 8fd13ca9b..d011c8983 100644 --- a/services/core/internal/db/sqlc/models.go +++ b/services/core/internal/db/sqlc/models.go @@ -342,6 +342,7 @@ type RuntimeNodeGenerationStatus struct { State string `json:"state"` Diagnostic string `json:"diagnostic"` ObservedAt pgtype.Timestamptz `json:"observed_at"` + Checkpoint []byte `json:"checkpoint"` } type RuntimePlacement struct { diff --git a/services/core/internal/db/sqlc/node_generations.sql.go b/services/core/internal/db/sqlc/node_generations.sql.go index ec512ca60..f52041660 100644 --- a/services/core/internal/db/sqlc/node_generations.sql.go +++ b/services/core/internal/db/sqlc/node_generations.sql.go @@ -43,6 +43,42 @@ func (q *Queries) GetNodeGenerationSpecification(ctx context.Context, generation return i, err } +const listCheckpointGenerationNodes = `-- name: ListCheckpointGenerationNodes :many +SELECT n.id, g.checkpoint +FROM runtime_nodes n +JOIN runtime_node_generation_status g ON g.node_id=n.id +CROSS JOIN runtime_deployment d +WHERE g.generation=$1 AND n.installation_id=d.installation_id AND n.removed_at IS NULL +AND n.connection_id=g.connection_id AND n.connected_epoch=g.owner_epoch AND g.owner_epoch=d.owner_epoch +AND n.last_seen_at>clock_timestamp()-interval '45 seconds' AND g.state='ready' +ORDER BY n.id +` + +type ListCheckpointGenerationNodesRow struct { + ID pgtype.UUID `json:"id"` + Checkpoint []byte `json:"checkpoint"` +} + +func (q *Queries) ListCheckpointGenerationNodes(ctx context.Context, generation int64) ([]ListCheckpointGenerationNodesRow, error) { + rows, err := q.db.Query(ctx, listCheckpointGenerationNodes, generation) + if err != nil { + return nil, err + } + defer rows.Close() + items := []ListCheckpointGenerationNodesRow{} + for rows.Next() { + var i ListCheckpointGenerationNodesRow + if err := rows.Scan(&i.ID, &i.Checkpoint); err != nil { + return nil, err + } + items = append(items, i) + } + if err := rows.Err(); err != nil { + return nil, err + } + return items, nil +} + const nodeGenerationKept = `-- name: NodeGenerationKept :one SELECT (EXISTS(SELECT 1 FROM runtime_deployment d WHERE d.generation=$1) OR EXISTS(SELECT 1 FROM runtime_nodes n WHERE n.id=$2 AND n.removed_at IS NULL AND n.ready_generation=$1) @@ -114,10 +150,10 @@ func (q *Queries) RefreshNodeServingReadiness(ctx context.Context, arg RefreshNo } const upsertNodeGenerationStatus = `-- name: UpsertNodeGenerationStatus :exec -INSERT INTO runtime_node_generation_status(node_id,generation,specification_digest,connection_id,owner_epoch,state,diagnostic) -VALUES($1,$2,$3,$4,$5,$6,$7) +INSERT INTO runtime_node_generation_status(node_id,generation,specification_digest,connection_id,owner_epoch,state,diagnostic,checkpoint) +VALUES($1,$2,$3,$4,$5,$6,$7,$8) ON CONFLICT(node_id,generation) DO UPDATE SET specification_digest=EXCLUDED.specification_digest,connection_id=EXCLUDED.connection_id, - owner_epoch=EXCLUDED.owner_epoch,state=EXCLUDED.state,diagnostic=EXCLUDED.diagnostic,observed_at=clock_timestamp() + owner_epoch=EXCLUDED.owner_epoch,state=EXCLUDED.state,diagnostic=EXCLUDED.diagnostic,checkpoint=EXCLUDED.checkpoint,observed_at=clock_timestamp() ` type UpsertNodeGenerationStatusParams struct { @@ -128,6 +164,7 @@ type UpsertNodeGenerationStatusParams struct { OwnerEpoch int64 `json:"owner_epoch"` State string `json:"state"` Diagnostic string `json:"diagnostic"` + Checkpoint []byte `json:"checkpoint"` } func (q *Queries) UpsertNodeGenerationStatus(ctx context.Context, arg UpsertNodeGenerationStatusParams) error { @@ -139,6 +176,7 @@ func (q *Queries) UpsertNodeGenerationStatus(ctx context.Context, arg UpsertNode arg.OwnerEpoch, arg.State, arg.Diagnostic, + arg.Checkpoint, ) return err } diff --git a/services/core/internal/db/sqlc/runtime_suspension.sql.go b/services/core/internal/db/sqlc/runtime_suspension.sql.go index 64fba0e08..9596b297e 100644 --- a/services/core/internal/db/sqlc/runtime_suspension.sql.go +++ b/services/core/internal/db/sqlc/runtime_suspension.sql.go @@ -61,6 +61,70 @@ func (q *Queries) GetRuntimeActivity(ctx context.Context, id pgtype.UUID) (GetRu return i, err } +const moveSuspendedRuntimeCompute = `-- name: MoveSuspendedRuntimeCompute :one +WITH moved AS ( + UPDATE runtime_placements p SET node_id=$1::uuid + FROM runtime_allocations a + WHERE a.id=$4 AND a.environment_id=p.environment_id + AND p.node_id=$5::uuid AND p.released_at IS NULL + AND p.deployment_generation=a.deployment_generation + AND a.node_id=p.node_id AND a.compute_revision=$6 + AND a.compute_phase='suspended' + AND ((a.state='running' AND $2::text='restoring' + AND a.compute_retained_until>clock_timestamp() + AND EXISTS(SELECT 1 FROM environments e WHERE e.id=a.environment_id AND e.initialization='complete')) + OR (a.state='cleanup_pending' AND $2::text='suspended')) + RETURNING a.id +) +UPDATE runtime_allocations a +SET node_id=$1::uuid, compute_phase=$2, + compute_state=$3::jsonb, compute_revision=a.compute_revision+1, + compute_phase_changed_at=CASE WHEN a.compute_phase=$2::text THEN a.compute_phase_changed_at ELSE clock_timestamp() END +FROM moved WHERE a.id=moved.id RETURNING a.id, a.environment_id, a.device_id, a.provider_key, a.state, a.create_settled, a.created_at, a.released_at, a.compute_phase, a.compute_revision, a.compute_state, a.compute_activity_at, a.compute_wake_requested, a.compute_retained_until, a.node_id, a.observation_error, a.compute_phase_changed_at, a.deployment_generation +` + +type MoveSuspendedRuntimeComputeParams struct { + Destination pgtype.UUID `json:"destination"` + Phase string `json:"phase"` + State []byte `json:"state"` + ID pgtype.UUID `json:"id"` + Source pgtype.UUID `json:"source"` + Revision int64 `json:"revision"` +} + +func (q *Queries) MoveSuspendedRuntimeCompute(ctx context.Context, arg MoveSuspendedRuntimeComputeParams) (RuntimeAllocation, error) { + row := q.db.QueryRow(ctx, moveSuspendedRuntimeCompute, + arg.Destination, + arg.Phase, + arg.State, + arg.ID, + arg.Source, + arg.Revision, + ) + var i RuntimeAllocation + err := row.Scan( + &i.ID, + &i.EnvironmentID, + &i.DeviceID, + &i.ProviderKey, + &i.State, + &i.CreateSettled, + &i.CreatedAt, + &i.ReleasedAt, + &i.ComputePhase, + &i.ComputeRevision, + &i.ComputeState, + &i.ComputeActivityAt, + &i.ComputeWakeRequested, + &i.ComputeRetainedUntil, + &i.NodeID, + &i.ObservationError, + &i.ComputePhaseChangedAt, + &i.DeploymentGeneration, + ) + return i, err +} + const recordRuntimeTerminalActivity = `-- name: RecordRuntimeTerminalActivity :exec UPDATE runtime_allocations a SET compute_activity_at = clock_timestamp() FROM environments e diff --git a/services/core/internal/db/sqlc/runtime_suspension_pressure.sql.go b/services/core/internal/db/sqlc/runtime_suspension_pressure.sql.go index da19d0c89..e192c879e 100644 --- a/services/core/internal/db/sqlc/runtime_suspension_pressure.sql.go +++ b/services/core/internal/db/sqlc/runtime_suspension_pressure.sql.go @@ -39,33 +39,34 @@ func (q *Queries) ListRuntimeSuspensionInFlightNodes(ctx context.Context) ([]pgt return items, nil } -const listWaitingRuntimeRestoreGenerations = `-- name: ListWaitingRuntimeRestoreGenerations :many -SELECT DISTINCT a.deployment_generation -FROM runtime_allocations a -JOIN environments e ON e.id = a.environment_id -JOIN sessions s ON s.id = e.session_id -WHERE a.node_id = $1 AND a.state = 'running' AND a.compute_phase = 'suspended' - AND a.compute_retained_until > clock_timestamp() - AND s.deleted_at IS NULL AND e.status NOT IN ('failed', 'expired') - AND (a.compute_wake_requested OR EXISTS ( - SELECT 1 FROM environment_input_reservations r - WHERE r.session_id = s.id AND r.state = 'pending' AND r.deadline > clock_timestamp() - )) +const listWaitingCheckpointRestores = `-- name: ListWaitingCheckpointRestores :many +SELECT DISTINCT a.node_id, a.deployment_generation, (a.compute_state->'snapshot'->'Compatibility')::jsonb AS checkpoint +FROM runtime_allocations a JOIN environments e ON e.id=a.environment_id JOIN sessions s ON s.id=e.session_id +WHERE a.node_id IS NOT NULL AND a.state='running' AND a.compute_phase='suspended' + AND a.compute_retained_until>clock_timestamp() AND s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND a.compute_state->'snapshot'->'Compatibility' IS NOT NULL + AND (a.compute_wake_requested OR EXISTS(SELECT 1 FROM environment_input_reservations r WHERE r.session_id=s.id AND r.state='pending' AND r.deadline>clock_timestamp())) ` -func (q *Queries) ListWaitingRuntimeRestoreGenerations(ctx context.Context, nodeID pgtype.UUID) ([]pgtype.Int8, error) { - rows, err := q.db.Query(ctx, listWaitingRuntimeRestoreGenerations, nodeID) +type ListWaitingCheckpointRestoresRow struct { + NodeID pgtype.UUID `json:"node_id"` + DeploymentGeneration pgtype.Int8 `json:"deployment_generation"` + Checkpoint []byte `json:"checkpoint"` +} + +func (q *Queries) ListWaitingCheckpointRestores(ctx context.Context) ([]ListWaitingCheckpointRestoresRow, error) { + rows, err := q.db.Query(ctx, listWaitingCheckpointRestores) if err != nil { return nil, err } defer rows.Close() - items := []pgtype.Int8{} + items := []ListWaitingCheckpointRestoresRow{} for rows.Next() { - var deployment_generation pgtype.Int8 - if err := rows.Scan(&deployment_generation); err != nil { + var i ListWaitingCheckpointRestoresRow + if err := rows.Scan(&i.NodeID, &i.DeploymentGeneration, &i.Checkpoint); err != nil { return nil, err } - items = append(items, deployment_generation) + items = append(items, i) } if err := rows.Err(); err != nil { return nil, err diff --git a/services/core/internal/deployment/allocations.go b/services/core/internal/deployment/allocations.go index 8a1aa92ae..24560832a 100644 --- a/services/core/internal/deployment/allocations.go +++ b/services/core/internal/deployment/allocations.go @@ -226,7 +226,7 @@ func (e *ExecutionOperations) cleanup(ctx context.Context, owner Allocation, abs // running compute to quiescing requires idle compute and either the idle // timeout or capacity pressure, rechecked in the transaction. func (e *ExecutionOperations) SetCompute(ctx context.Context, owner Allocation, phase string, state json.RawMessage, retainedUntil *time.Time, idleTimeout time.Duration) (Allocation, error) { - if !computeTransition(owner.ComputePhase, phase) || !json.Valid(state) || (owner.ComputePhase == "running" && phase == "quiescing" && idleTimeout <= 0) { + if owner.ComputePhase == "suspended" && phase == "restoring" || !computeTransition(owner.ComputePhase, phase) || !json.Valid(state) || (owner.ComputePhase == "running" && phase == "quiescing" && idleTimeout <= 0) { return Allocation{}, ErrInvalidInput } if phase != "running" && (retainedUntil == nil || retainedUntil.IsZero()) { @@ -267,20 +267,79 @@ func (e *ExecutionOperations) SetCompute(ctx context.Context, owner Allocation, } } } - // A node-backed restore reserves capacity on its original node. - if current.ComputePhase == "suspended" && phase == "restoring" && current.NodeID != "" { - restore, err := tx.LoadRestore(current) - if err != nil { - return Allocation{}, err - } - if err := placement.CheckRestore(restore); err != nil { - return Allocation{}, err - } - } return tx.SetCompute(current, ComputeChange{Phase: phase, State: state, RetainedUntil: retainedUntil}) }) } +// BeginRestore reserves a compatible destination before any native restore. +// The caller supplies a fresh restore intent without a target; the destination +// adapter derives the target after this commit. Snapshot portability proves +// that publication and source-local cleanup settled before suspension. +func (e *ExecutionOperations) BeginRestore(ctx context.Context, owner Allocation, compatibility sandbox.CheckpointCompatibility, state json.RawMessage) (Allocation, error) { + var object map[string]json.RawMessage + if compatibility.Validate() != nil || json.Unmarshal(state, &object) != nil || object == nil { + return Allocation{}, ErrInvalidInput + } + return e.change(ctx, owner, true, func(tx AllocationTx, current Allocation) (Allocation, error) { + if current.State != "running" || current.ComputePhase != "suspended" || current.ComputeRevision != owner.ComputeRevision || current.Expired || current.ComputeRetainedUntil == nil { + return Allocation{}, ErrAllocationConflict + } + activity, err := tx.LoadActivity(current) + if err != nil { + return Allocation{}, err + } + if !activity.Busy && !activity.WakeRequested { + return Allocation{}, ErrAllocationConflict + } + if current.NodeID == "" { + return tx.SetCompute(current, ComputeChange{Phase: "restoring", State: state, RetainedUntil: current.ComputeRetainedUntil}) + } + d, nodes, err := tx.LoadCheckpointPlacement(current) + if err != nil { + return Allocation{}, err + } + if d.Resetting { + return Allocation{}, placement.ErrResetAdmission + } + if d.InstallationID != current.ProviderKey || d.Mode != string(sandbox.DeploymentNodes) { + return Allocation{}, placement.ErrNodeUnavailable + } + node, err := e.service.rules.ChooseCheckpoint(nodes, current.NodeID, compatibility, false) + if err != nil { + return Allocation{}, err + } + return tx.MoveSuspended(current, node, ComputeChange{Phase: "restoring", State: state, RetainedUntil: current.ComputeRetainedUntil}) + }) +} + +// RelocateCleanup durably assigns a portable suspended artifact to a node that +// can delete it. An unknown restore remains pinned to its existing target. +func (e *ExecutionOperations) RelocateCleanup(ctx context.Context, owner Allocation, compatibility sandbox.CheckpointCompatibility) (Allocation, error) { + if owner.NodeID == "" || compatibility.Validate() != nil { + return Allocation{}, ErrInvalidInput + } + return e.change(ctx, owner, false, func(tx AllocationTx, current Allocation) (Allocation, error) { + if current.State != "cleanup_pending" || current.ComputePhase != "suspended" || current.ComputeRevision != owner.ComputeRevision { + return Allocation{}, ErrAllocationConflict + } + d, nodes, err := tx.LoadCheckpointPlacement(current) + if err != nil { + return Allocation{}, err + } + if d.InstallationID != current.ProviderKey || d.Mode != string(sandbox.DeploymentNodes) { + return Allocation{}, placement.ErrNodeUnavailable + } + node, err := e.service.rules.ChooseCheckpoint(nodes, current.NodeID, compatibility, true) + if err != nil { + return Allocation{}, err + } + if node == current.NodeID { + return current, nil + } + return tx.MoveSuspended(current, node, ComputeChange{Phase: current.ComputePhase, State: current.ComputeState, RetainedUntil: current.ComputeRetainedUntil}) + }) +} + // RecordObservation records a diagnostic for the observed compute revision // and state of a node-backed allocation. It never releases ownership, changes // public readiness or authorizes replacement. diff --git a/services/core/internal/deployment/allocations_test.go b/services/core/internal/deployment/allocations_test.go index 4de111df1..22d110171 100644 --- a/services/core/internal/deployment/allocations_test.go +++ b/services/core/internal/deployment/allocations_test.go @@ -15,6 +15,7 @@ import ( "github.com/google/uuid" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" ) @@ -141,9 +142,13 @@ func (f *fakeAllocationTx) LoadGenerationSpecification(uint64) (GenerationSpecif return GenerationSpecification{}, nil } -func (f *fakeAllocationTx) LoadRestore(Allocation) (placement.Restore, error) { - unexpected(f.t, "LoadRestore") - return placement.Restore{}, nil +func (f *fakeAllocationTx) LoadCheckpointPlacement(Allocation) (placement.Deployment, []placement.CheckpointNode, error) { + unexpected(f.t, "LoadCheckpointPlacement") + return placement.Deployment{}, nil, nil +} +func (f *fakeAllocationTx) MoveSuspended(Allocation, string, ComputeChange) (Allocation, error) { + unexpected(f.t, "MoveSuspended") + return Allocation{}, nil } func (f *fakeAllocationTx) ObserveRunning(Allocation) (Allocation, error) { @@ -630,7 +635,13 @@ func TestSetComputeRechecksIdlePressureAndActivity(t *testing.T) { }, loadSuspensionDemand: func(Allocation) (SuspensionDemand, error) { pressureChecks++ - return SuspensionDemand{Deployment: placement.Deployment{InstallationID: owner.ProviderKey, Mode: "nodes"}, Nodes: []placement.Node{{ID: owner.NodeID, Online: true, Active: 1, MaxActive: 1}}, RestoreWaiting: test.pressure}, nil + node := placement.Node{ID: owner.NodeID, Online: true, Active: 1, MaxActive: 1, Retained: 1, MaxRetained: 1, CoreURL: testPublicURL} + demand := SuspensionDemand{Deployment: placement.Deployment{InstallationID: owner.ProviderKey, Mode: "nodes"}, Nodes: []placement.Node{node}} + if test.pressure { + compatibility := sandbox.CheckpointCompatibility{ArtifactDomain: "fixture", ExecutionClass: "fixture"} + demand.CheckpointRestores = []CheckpointRestoreDemand{{SourceNode: owner.NodeID, Compatibility: compatibility, Nodes: []placement.CheckpointNode{{Node: node, Checkpoint: &compatibility}}}} + } + return demand, nil }, setCompute: func(a Allocation, c ComputeChange) (Allocation, error) { writes++ @@ -661,3 +672,29 @@ func (f *fakeAllocationTx) CanRetainEnvironment(current Allocation) (bool, error } return f.canRetainEnvironment(current) } + +func TestBeginRestoreDirectKeepsAllocationAndRetention(t *testing.T) { + until := time.Now().Add(time.Hour) + owner := Allocation{ID: uuid.NewString(), DeviceID: uuid.NewString(), TenantID: uuid.NewString(), EnvironmentID: uuid.NewString(), ProviderKey: uuid.NewString(), State: "running", ComputePhase: "suspended", ComputeRetainedUntil: &until} + tx := &fakeAllocationTx{t: t, loadAllocation: func() (Allocation, error) { return owner, nil }, loadSessionDevice: func() (SessionDevice, bool, error) { + return SessionDevice{ID: owner.DeviceID, EnvironmentID: owner.EnvironmentID}, true, nil + }, loadActivity: func(Allocation) (Activity, error) { return Activity{WakeRequested: true}, nil }, setCompute: func(current Allocation, change ComputeChange) (Allocation, error) { + if change.Phase != "restoring" || change.RetainedUntil != owner.ComputeRetainedUntil { + t.Fatal("direct restore lost intent", change) + } + current.ComputePhase = change.Phase + return current, nil + }} + storage := &fakeExecutionStorage{t: t, withAllocation: func(_ context.Context, _ AllocationKey, apply func(AllocationTx) error) error { return apply(tx) }} + operations, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, testPublicURL), storage, engine.Catalog{}) + if err != nil { + t.Fatal(err) + } + restored, err := operations.BeginRestore(t.Context(), owner, sandbox.CheckpointCompatibility{ArtifactDomain: "fixture", ExecutionClass: "fixture"}, json.RawMessage(`{"restore_id":"attempt"}`)) + if err != nil || restored.ID != owner.ID || restored.NodeID != "" || restored.ComputePhase != "restoring" { + t.Fatal("direct restore changed ownership", restored, err) + } + if _, err := operations.SetCompute(t.Context(), owner, "restoring", json.RawMessage(`{}`), &until, time.Minute); !errors.Is(err, ErrInvalidInput) { + t.Fatal("restore bypassed capacity admission", err) + } +} diff --git a/services/core/internal/deployment/nodes.go b/services/core/internal/deployment/nodes.go index f166a12b8..94c281f5b 100644 --- a/services/core/internal/deployment/nodes.go +++ b/services/core/internal/deployment/nodes.go @@ -11,6 +11,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/coremetrics" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/providercontract" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/google/uuid" ) @@ -516,8 +517,8 @@ func (s *Service) NodeRetention(ctx context.Context, nodeID, connectionID string // Heartbeat records a protocol 1 node's health, which reports readiness of its // enrollment generation only. -func (s *Service) Heartbeat(ctx context.Context, nodeID, connectionID string, epoch uint64, health NodeHealth) error { - return s.heartbeat(ctx, nodeID, connectionID, epoch, health, nil, 1) +func (s *Service) Heartbeat(ctx context.Context, nodeID, connectionID string, epoch uint64, health NodeHealth, statuses []sandbox.GenerationStatus) error { + return s.heartbeat(ctx, nodeID, connectionID, epoch, health, statuses, 1) } // HeartbeatGenerations records a node's health and the preparation state of @@ -558,7 +559,12 @@ func (s *Service) heartbeat(ctx context.Context, nodeID, connectionID string, ep if !current { return ErrNodeCredential } - if protocol == 1 { + if protocol == 1 && len(statuses) > 0 { + if len(statuses) != 1 || statuses[0].Generation != n.DeploymentGeneration || statuses[0].SpecificationDigest != n.SpecificationDigest || (statuses[0].State == "ready") != health.ProviderReady { + return ErrInvalidInput + } + } + if protocol == 1 && len(statuses) == 0 { state := "failed" if health.ProviderReady { state = "ready" @@ -570,8 +576,16 @@ func (s *Service) heartbeat(ctx context.Context, nodeID, connectionID string, ep } func (s *Service) recordGenerations(tx NodeTx, d Record, n StoredNode, statuses []sandbox.GenerationStatus, protocol int) error { + adapter, err := s.registry.Lookup(d.Provider) + if err != nil { + return err + } + checkpointSupported := adapter.Operations()["Initial"].State == providercontract.Supported seen := map[uint64]bool{} for _, status := range statuses { + if status.Checkpoint != nil && (status.State != "ready" || !checkpointSupported || status.Checkpoint.Validate() != nil) || status.State == "ready" && checkpointSupported && status.Checkpoint == nil { + return ErrInvalidInput + } if !validGeneration(status.Generation) || !validDigest(status.SpecificationDigest) || seen[status.Generation] || (status.State != "ready" && status.State != "preparing" && status.State != "failed") || status.State == "ready" && status.Diagnostic != "" { return ErrInvalidInput } @@ -592,7 +606,7 @@ func (s *Service) recordGenerations(tx NodeTx, d Record, n StoredNode, statuses if spec.Digest(d.Provider) != status.SpecificationDigest { return ErrSpecificationMismatch } - if err := tx.UpsertGenerationStatus(GenerationStatusRecord{NodeID: n.ID, ConnectionID: n.ConnectionID, Generation: status.Generation, SpecificationDigest: status.SpecificationDigest, OwnerEpoch: d.OwnerEpoch, State: status.State, Diagnostic: sandbox.NormalizeNodeDiagnostic(status.Diagnostic)}); err != nil { + if err := tx.UpsertGenerationStatus(GenerationStatusRecord{NodeID: n.ID, ConnectionID: n.ConnectionID, Generation: status.Generation, SpecificationDigest: status.SpecificationDigest, OwnerEpoch: d.OwnerEpoch, State: status.State, Diagnostic: sandbox.NormalizeNodeDiagnostic(status.Diagnostic), Checkpoint: status.Checkpoint}); err != nil { return err } if status.State == "ready" && status.Generation == d.Generation { diff --git a/services/core/internal/deployment/placement/placement.go b/services/core/internal/deployment/placement/placement.go index d68b8e153..151deea6a 100644 --- a/services/core/internal/deployment/placement/placement.go +++ b/services/core/internal/deployment/placement/placement.go @@ -109,14 +109,41 @@ type Reserved struct { Available bool } -// Restore is what restoring suspended compute on its node reads: the node, -// nil when the installation no longer has it, and whether the node is ready -// for the allocation's generation. -type Restore struct { - Node *Node - // Generation is zero when the allocation has none. - Generation uint64 - GenerationReady bool +// CheckpointNode is a fresh readiness declaration for one immutable generation. +// Capacity is read under the same deployment lock as the allocation transfer. +type CheckpointNode struct { + Node Node + Checkpoint *sandbox.CheckpointCompatibility +} + +// ChooseCheckpoint selects a compatible node without interpreting adapter tokens. +// A retained allocation already owns its source slot. Cleanup needs no active +// slot and compares only the artifact domain, but still reserves a retained slot. +func (r *Rules) ChooseCheckpoint(nodes []CheckpointNode, source string, compatibility sandbox.CheckpointCompatibility, cleanup bool) (string, error) { + var chosen *Node + for i := range nodes { + candidate := &nodes[i] + n := &candidate.Node + if !n.Online || candidate.Checkpoint == nil || candidate.Checkpoint.ArtifactDomain != compatibility.ArtifactDomain { + continue + } + if !cleanup && (candidate.Checkpoint.ExecutionClass != compatibility.ExecutionClass || n.CoreURL != r.publicURL || n.Active >= int64(n.MaxActive)) { + continue + } + if n.ID != source && n.Retained >= int64(n.MaxRetained) { + continue + } + if n.ID == source { + return n.ID, nil + } + if chosen == nil || n.Active < chosen.Active || n.Active == chosen.Active && n.Retained < chosen.Retained { + chosen = n + } + } + if chosen == nil { + return "", ErrNodeUnavailable + } + return chosen.ID, nil } // CheckPublicOrigin rejects a provider that requires a reachable public @@ -205,17 +232,6 @@ func CheckReserved(reserved Reserved) error { return nil } -// CheckRestore admits restoring suspended compute on its original node: the -// node must be online, ready for the allocation's generation and below its -// active capacity. Restore never selects another node. -func CheckRestore(restore Restore) error { - n := restore.Node - if n == nil || restore.Generation == 0 || !n.Online || !restore.GenerationReady || n.Active >= int64(n.MaxActive) { - return ErrNodeUnavailable - } - return nil -} - // LoopbackOrigin reports whether a validated origin names a loopback host, // which nothing outside the Core host can reach. func LoopbackOrigin(value string) bool { diff --git a/services/core/internal/deployment/placement/placement_test.go b/services/core/internal/deployment/placement/placement_test.go index 473c343bb..ae9c7442f 100644 --- a/services/core/internal/deployment/placement/placement_test.go +++ b/services/core/internal/deployment/placement/placement_test.go @@ -158,7 +158,7 @@ func TestDecidePlacement(t *testing.T) { } } -func TestCheckReservedAndRestore(t *testing.T) { +func TestCheckReserved(t *testing.T) { for name, test := range map[string]struct { reserved Reserved want error @@ -171,24 +171,6 @@ func TestCheckReservedAndRestore(t *testing.T) { t.Errorf("%s: CheckReserved = %v, want %v", name, err, test.want) } } - node := func(online bool, active int64) *Node { - return &Node{ID: "a", Online: online, Active: active, MaxActive: 2} - } - for name, test := range map[string]struct { - restore Restore - want error - }{ - "ready": {Restore{Node: node(true, 1), Generation: 3, GenerationReady: true}, nil}, - "missing node": {Restore{Generation: 3, GenerationReady: true}, ErrNodeUnavailable}, - "no generation": {Restore{Node: node(true, 1), GenerationReady: true}, ErrNodeUnavailable}, - "offline": {Restore{Node: node(false, 1), Generation: 3, GenerationReady: true}, ErrNodeUnavailable}, - "generation unready": {Restore{Node: node(true, 1), Generation: 3}, ErrNodeUnavailable}, - "at capacity": {Restore{Node: node(true, 2), Generation: 3, GenerationReady: true}, ErrNodeUnavailable}, - } { - if err := CheckRestore(test.restore); !errors.Is(err, test.want) || (test.want == nil) != (err == nil) { - t.Errorf("%s: CheckRestore = %v, want %v", name, err, test.want) - } - } } func TestLoopbackOrigin(t *testing.T) { diff --git a/services/core/internal/deployment/storage.go b/services/core/internal/deployment/storage.go index d25af71cf..5277a9c3a 100644 --- a/services/core/internal/deployment/storage.go +++ b/services/core/internal/deployment/storage.go @@ -128,9 +128,12 @@ type AllocationTx interface { PlacementDemand(after PlacementDemandCursor) ([]PlacementDemand, PlacementDemandCursor, error) // LoadGenerationSpecification reads immutable generation requirements. LoadGenerationSpecification(generation uint64) (GenerationSpecification, error) - // LoadRestore locks the deployment and returns what restoring the - // allocation's suspended compute on its node reads. - LoadRestore(current Allocation) (placement.Restore, error) + // LoadCheckpointPlacement locks the deployment and loads fresh readiness + // for the allocation's exact immutable generation, including capacity. + LoadCheckpointPlacement(current Allocation) (placement.Deployment, []placement.CheckpointNode, error) + // MoveSuspended changes node ownership, placement and compute intent in one + // transaction. Only a settled suspended allocation can change nodes. + MoveSuspended(current Allocation, nodeID string, change ComputeChange) (Allocation, error) // ObserveRunning records the allocation running with its creation settled. ObserveRunning(current Allocation) (Allocation, error) // SettleCreation records that the original Create can no longer change @@ -506,4 +509,5 @@ type GenerationStatusRecord struct { SpecificationDigest string OwnerEpoch uint64 State, Diagnostic string + Checkpoint *sandbox.CheckpointCompatibility } diff --git a/services/core/internal/deployment/suspension_pressure.go b/services/core/internal/deployment/suspension_pressure.go index 06c0bf492..d8ae4c173 100644 --- a/services/core/internal/deployment/suspension_pressure.go +++ b/services/core/internal/deployment/suspension_pressure.go @@ -11,10 +11,17 @@ import ( // SuspensionDemand is a transaction-bound capacity observation. It is not a // reservation: only a confirmed suspension frees an active slot. type SuspensionDemand struct { - Deployment placement.Deployment - Nodes []placement.Node - InFlightNodes []string - RestoreWaiting bool + Deployment placement.Deployment + Nodes []placement.Node + InFlightNodes []string + CheckpointRestores []CheckpointRestoreDemand +} + +// CheckpointRestoreDemand carries one stopped owner and exact-generation target evidence. +type CheckpointRestoreDemand struct { + SourceNode string + Compatibility sandbox.CheckpointCompatibility + Nodes []placement.CheckpointNode } func (e *ExecutionOperations) canSuspendForDemand(tx AllocationTx, owner Allocation) (bool, error) { @@ -35,10 +42,36 @@ func (e *ExecutionOperations) canSuspendForDemand(tx AllocationTx, owner Allocat if source == nil || !source.Online || source.Active < int64(source.MaxActive) { return false, nil } - // A retained restore already owns its retained slot and may use an older - // prepared generation than the node's serving generation. - if pressure.RestoreWaiting { - return !slices.Contains(pressure.InFlightNodes, owner.NodeID), nil + for _, demand := range pressure.CheckpointRestores { + if _, err := e.service.rules.ChooseCheckpoint(demand.Nodes, demand.SourceNode, demand.Compatibility, false); err == nil { + continue + } else if !errors.Is(err, placement.ErrNodeUnavailable) { + return false, err + } + candidates := slices.Clone(demand.Nodes) + for i := range candidates { + if candidates[i].Node.ID == owner.NodeID { + candidates[i].Node.Active-- + } + } + chosen, err := e.service.rules.ChooseCheckpoint(candidates, demand.SourceNode, demand.Compatibility, false) + if err == nil && chosen == owner.NodeID { + eligible := make([]placement.Node, 0, len(candidates)) + for _, candidate := range candidates { + if candidate.Checkpoint != nil && *candidate.Checkpoint == demand.Compatibility { + eligible = append(eligible, candidate.Node) + } + } + inFlight, err := e.reclamationInFlight(pressure, eligible) + if err != nil { + return false, err + } + if !inFlight { + return true, nil + } + } else if err != nil && !errors.Is(err, placement.ErrNodeUnavailable) { + return false, err + } } if source.Retained >= int64(source.MaxRetained) { return false, nil diff --git a/services/core/internal/execution/runtime_compute.go b/services/core/internal/execution/runtime_compute.go index d6f2f62eb..3b97738e4 100644 --- a/services/core/internal/execution/runtime_compute.go +++ b/services/core/internal/execution/runtime_compute.go @@ -199,6 +199,9 @@ func (r *runtimeLifecycle) captureCompute(ctx context.Context, p sandbox.Sandbox } return r.wakeCompute(ctx, p, next, state) } + if result.Snapshot.Compatibility.Validate() != nil { + return sandbox.ErrOwnership + } state.Snapshot = result.Snapshot // Store the verified artifact before any recovery-path kill. Snapshot failure // or an unknown result cannot silently fall back to a cold Environment. @@ -207,8 +210,9 @@ func (r *runtimeLifecycle) captureCompute(ctx context.Context, p sandbox.Sandbox return err } owner = next - if err := ignoreComputeAbsent(p.KillCompute(ctx, runtimeReference(owner), state.Current)); err != nil { - return err + if !result.SourceStopped { + // Native absence alone does not settle capture or publish the archive. + return sandbox.ErrComputeUnconfirmed } _, err = r.saveCompute(ctx, next, "suspended", state, next.ComputeRetainedUntil) return err @@ -226,22 +230,40 @@ func (r *runtimeLifecycle) restoreIdleCompute(ctx context.Context, p sandbox.San if state.Snapshot == nil || state.Target != nil { return sandbox.ErrOwnership } - target, err := p.NewCompute(ctx, runtimeReference(owner), state.Current.Generation+1, state.Snapshot) + state.RestoreID = uuid.NewString() + raw, err := json.Marshal(state) if err != nil { return err } - state.Target, state.RestoreID = &target, uuid.NewString() - next, err := r.saveCompute(ctx, owner, "restoring", state, owner.ComputeRetainedUntil) + next, err := r.deployment.BeginRestore(ctx, owner, state.Snapshot.Compatibility, raw) if err != nil { return err } + if next.NodeID != r.nodeID { + // The durable route and reservation belong to the target node's lane. + return nil + } return r.restoreCompute(ctx, p, next, state, false) } func (r *runtimeLifecycle) restoreCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute, observeOnly bool) (err error) { defer func() { err = withObservationOwner(owner, err) }() - if state.Target == nil || state.Snapshot == nil || state.Rollback { + if state.Snapshot == nil || state.Snapshot.Compatibility.Validate() != nil || state.Rollback || state.RestoreID == "" { return sandbox.ErrOwnership } + if state.Target == nil { + // NewCompute derives identity without starting native execution. A crash + // before this commit can safely repeat derivation on the reserved target. + target, err := p.NewCompute(ctx, runtimeReference(owner), state.Current.Generation+1, state.Snapshot) + if err != nil { + return err + } + state.Target = &target + next, err := r.saveCompute(ctx, owner, "restoring", state, owner.ComputeRetainedUntil) + if err != nil { + return err + } + owner, observeOnly = next, false + } spec, err := r.workspaceSpecification(ctx, owner.Key(), owner.ID) if err != nil { return err @@ -260,6 +282,18 @@ func (r *runtimeLifecycle) restoreCompute(ctx context.Context, p sandbox.Sandbox if err != nil { return err } + if result.RestoreAttemptClosed != "" { + if !result.ClosesRestoreAttempt(sandbox.ResumeRequest{OperationID: state.RestoreID, Snapshot: *state.Snapshot, Target: *state.Target, ObserveOnly: observeOnly}) { + return sandbox.ErrOwnership + } + // Only an exact, durably closed never-admitted attempt can be replaced. + state.RestoreID = uuid.NewString() + next, err := r.saveCompute(ctx, owner, "restoring", state, owner.ComputeRetainedUntil) + if err != nil { + return err + } + return r.restoreCompute(ctx, p, next, state, false) + } if result.Status != "running" || result.Compute.ID == "" { return sandbox.ErrComputeUnconfirmed } diff --git a/services/core/internal/execution/runtime_compute_wake.go b/services/core/internal/execution/runtime_compute_wake.go index 2aab604f6..bf16b5fb2 100644 --- a/services/core/internal/execution/runtime_compute_wake.go +++ b/services/core/internal/execution/runtime_compute_wake.go @@ -78,6 +78,16 @@ func (r *runtimeLifecycle) cleanupCompute(ctx context.Context, p sandbox.Sandbox if err := r.lease.CheckOwnership(ctx); err != nil { return err } + if owner.ComputePhase == "suspended" && state.Snapshot != nil && owner.NodeID != "" { + next, err := r.deployment.RelocateCleanup(ctx, owner, state.Snapshot.Compatibility) + if err != nil { + return err + } + if next.NodeID != r.nodeID { + return nil + } + owner = next + } // An uncommitted artifact is found by its persisted attempt, never a directory // glob. The helper's allocation lock also waits for an earlier unknown call. if owner.ComputePhase == "suspending" && state.Snapshot == nil { @@ -94,8 +104,10 @@ func (r *runtimeLifecycle) cleanupCompute(ctx context.Context, p sandbox.Sandbox return err } } - if err := ignoreComputeAbsent(p.KillCompute(ctx, runtimeReference(owner), state.Current)); err != nil { - return err + if owner.ComputePhase != "suspended" && owner.ComputePhase != "restoring" { + if err := ignoreComputeAbsent(p.KillCompute(ctx, runtimeReference(owner), state.Current)); err != nil { + return err + } } if state.Snapshot != nil { if err := ignoreComputeAbsent(p.DeleteSnapshot(ctx, runtimeReference(owner), *state.Snapshot)); err != nil { diff --git a/services/core/internal/execution/runtime_observation_test.go b/services/core/internal/execution/runtime_observation_test.go index 8ad10bdae..cc13c2f0c 100644 --- a/services/core/internal/execution/runtime_observation_test.go +++ b/services/core/internal/execution/runtime_observation_test.go @@ -18,7 +18,7 @@ func (p *observationCaptureProvider) Suspend(_ context.Context, q sandbox.Suspen if p.captureError != nil { return sandbox.ComputeState{}, p.captureError } - return sandbox.ComputeState{Compute: q.Source, Snapshot: &sandbox.SnapshotIdentity{ID: "captured"}, Status: "suspended", SourceStopped: true}, nil + return sandbox.ComputeState{Compute: q.Source, Snapshot: &sandbox.SnapshotIdentity{ID: "captured", Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}}, Status: "suspended", SourceStopped: p.killError == nil}, nil } func TestComputeFailureObservationUsesCommittedReceipt(t *testing.T) { diff --git a/services/core/internal/execution/runtime_replacement_test.go b/services/core/internal/execution/runtime_replacement_test.go index 9a4e2f098..1d8c2c31f 100644 --- a/services/core/internal/execution/runtime_replacement_test.go +++ b/services/core/internal/execution/runtime_replacement_test.go @@ -132,9 +132,9 @@ func TestExpiredRetainedComputePreservesSessionAndPendingInput(t *testing.T) { if _, err = fixture.pool.Exec(t.Context(), `UPDATE session_devices SET native_session_id='retained-native-session' WHERE device_id=$1`, owner.DeviceID); err != nil { t.Fatal(err) } - compute := runtimeCompute{Current: sandbox.Compute{ID: "old-compute"}, Snapshot: &sandbox.SnapshotIdentity{ID: "old-snapshot"}} + compute := runtimeCompute{Current: sandbox.Compute{ID: "old-compute"}, Snapshot: &sandbox.SnapshotIdentity{Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, ID: "old-snapshot"}} raw, _ := json.Marshal(compute) - if _, err = fixture.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspended',compute_state=$2,compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID, raw); err != nil { + if _, err = fixture.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspending',compute_state=$2,compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID, raw); err != nil { t.Fatal(err) } input, err := fixture.sessions.ReserveEnvironmentInput(t.Context(), session.TenantID, session.ID, "after-retention", []sessions.Input{{Kind: "message", Payload: json.RawMessage(`{"input":[{"role":"user","content":[{"type":"input_text","text":"resume"}]}]}`)}}) @@ -213,7 +213,7 @@ func TestRetainedEnvironmentDemandRecreatesCompute(t *testing.T) { {`UPDATE environments SET initialization='complete',status='disconnected' WHERE id=$1`, owner.EnvironmentID}, {`UPDATE devices SET supported_agent_kinds='[{"kind":"codex","available":true,"capabilities":{"retained_native_history":true}}]' WHERE id=$1`, owner.DeviceID}, {`UPDATE session_devices SET native_session_id='retained-native-session' WHERE device_id=$1`, owner.DeviceID}, - {`UPDATE runtime_allocations SET compute_phase='suspended',compute_state='{"current":{"id":"old-compute"},"snapshot":{"id":"old-snapshot"}}',compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID}, + {`UPDATE runtime_allocations SET compute_phase='suspended',compute_state='{"current":{"id":"old-compute"},"snapshot":{"id":"old-snapshot","Compatibility":{"artifact_domain":"fixture-store","execution_class":"fixture-runtime"}}}',compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID}, } { if _, err = fixture.pool.Exec(t.Context(), statement.q, statement.arg); err != nil { t.Fatal(err) @@ -354,7 +354,7 @@ func workspaceNodeSetup(t *testing.T, external bool) func(Owner, *deployment.Ser if err = service.ConnectNode(t.Context(), node.NodeID, connection, epoch); err != nil { t.Fatal(err) } - if err = service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err = service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}, []sandbox.GenerationStatus{{Generation: view.Generation, SpecificationDigest: view.SpecificationDigest, State: "ready", Checkpoint: &sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}}}); err != nil { t.Fatal(err) } return installation, node.NodeID @@ -462,7 +462,7 @@ func TestRestoreUsesAllocationGenerationStorageInsteadOfCachedLane(t *testing.T) } r.config.Workspace = target.Workspace r.config.Resources = target.Resources - state := runtimeCompute{Target: &sandbox.Compute{ID: "replacement-compute", Generation: 2}, Snapshot: &sandbox.SnapshotIdentity{ID: "snapshot"}, RestoreID: uuid.NewString()} + state := runtimeCompute{Target: &sandbox.Compute{ID: "replacement-compute", Generation: 2}, Snapshot: &sandbox.SnapshotIdentity{Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, ID: "snapshot"}, RestoreID: uuid.NewString()} if err = r.restoreCompute(t.Context(), provider, owner, state, false); !errors.Is(err, stop) { t.Fatal(err) } diff --git a/services/core/internal/execution/runtime_restore_attempt_test.go b/services/core/internal/execution/runtime_restore_attempt_test.go new file mode 100644 index 000000000..304811277 --- /dev/null +++ b/services/core/internal/execution/runtime_restore_attempt_test.go @@ -0,0 +1,227 @@ +package execution + +import ( + "context" + "encoding/json" + "errors" + "strings" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/google/uuid" +) + +type restoreAttemptProvider struct { + retentionProvider + outcome string + requests []sandbox.ResumeRequest + derived int + stop error +} + +func (p *restoreAttemptProvider) NewCompute(_ context.Context, _ sandbox.Reference, generation uint64, snapshot *sandbox.SnapshotIdentity) (sandbox.Compute, error) { + p.derived++ + return sandbox.Compute{Generation: generation, Name: "target", RestoredFrom: snapshot}, nil +} +func (p *restoreAttemptProvider) Resume(_ context.Context, q sandbox.ResumeRequest) (sandbox.ComputeState, error) { + p.requests = append(p.requests, q) + if !q.ObserveOnly { + return sandbox.ComputeState{}, p.stop + } + if p.outcome == "unknown" { + return sandbox.ComputeState{}, sandbox.ErrComputeUnconfirmed + } + closed := q.OperationID + if p.outcome == "wrong_operation" { + closed = uuid.NewString() + } + result := sandbox.ComputeState{Compute: q.Target, Status: "absent", RestoreAttemptClosed: closed} + if p.outcome == "wrong_target" { + result.Compute.Name = "another-target" + } + return result, nil +} + +func TestRestoreAttemptRecoveryUsesExactClosedEvidence(t *testing.T) { + for _, outcome := range []string{"unsent_identity", "closed", "unknown", "wrong_operation", "wrong_target"} { + t.Run(outcome, func(t *testing.T) { + stop := errors.New("native request captured") + provider := &restoreAttemptProvider{outcome: outcome, stop: stop} + f := workspaceSettlementFixtureWithSetup(t, provider, workspaceNodeSetup(t, true)) + r, session := f.lifecycle, f.session + owner, err := r.provision(t.Context(), session.TenantID, session.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + snapshot := &sandbox.SnapshotIdentity{ID: "snapshot", Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}} + state := runtimeCompute{Current: sandbox.Compute{ID: "source", Name: "source", Generation: 1}, Snapshot: snapshot, RestoreID: uuid.NewString()} + if outcome != "unsent_identity" { + state.Target = &sandbox.Compute{Name: "target", Generation: 2, RestoredFrom: snapshot} + } + raw, err := json.Marshal(state) + if err != nil { + t.Fatal(err) + } + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='restoring',compute_revision=4,compute_retained_until=clock_timestamp()+interval '1 hour',compute_state=$2 WHERE id=$1`, owner.ID, raw); err != nil { + t.Fatal(err) + } + owner, err = r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + err = r.restoreCompute(t.Context(), provider, owner, state, true) + current, readErr := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if readErr != nil { + t.Fatal(readErr) + } + var persisted runtimeCompute + if json.Unmarshal(current.ComputeState, &persisted) != nil { + t.Fatal("invalid persisted receipt") + } + switch outcome { + case "unsent_identity": + if !errors.Is(err, stop) || provider.derived != 1 || len(provider.requests) != 1 || provider.requests[0].ObserveOnly || persisted.Target == nil || persisted.RestoreID != state.RestoreID { + t.Fatalf("unsent intent not safely derived: %v %#v", err, provider.requests) + } + case "closed": + if !errors.Is(err, stop) || len(provider.requests) != 2 || !provider.requests[0].ObserveOnly || provider.requests[1].ObserveOnly || persisted.RestoreID == state.RestoreID || provider.requests[1].OperationID != persisted.RestoreID { + t.Fatalf("closed attempt not replaced durably: %v %#v", err, provider.requests) + } + default: + want := sandbox.ErrOwnership + if outcome == "unknown" { + want = sandbox.ErrComputeUnconfirmed + } + if !errors.Is(err, want) || len(provider.requests) != 1 || persisted.RestoreID != state.RestoreID || current.ComputeRevision != owner.ComputeRevision { + t.Fatalf("unproven attempt was replayed: %v %#v", err, provider.requests) + } + } + }) + } +} + +func TestCheckpointTransferRoutesBeforeTargetIO(t *testing.T) { + for _, operation := range []string{"wake", "expiry"} { + t.Run(operation, func(t *testing.T) { + stop := errors.New("target Resume captured") + provider := &restoreAttemptProvider{stop: stop} + f := workspaceSettlementFixtureWithSetup(t, provider, workspaceNodeSetup(t, true)) + r, session := f.lifecycle, f.session + owner, err := r.provision(t.Context(), session.TenantID, session.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + view, err := r.deployments.View(t.Context()) + if err != nil { + t.Fatal(err) + } + token, err := r.deployments.CreateEnrollment(t.Context(), deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + if err != nil { + t.Fatal(err) + } + targetNode := uuid.NewString() + node := deployment.Enrollment{NodeID: targetNode, Name: "checkpoint target", Credential: strings.Repeat("t", 64), Provider: view.Provider, BackendFingerprint: strings.Repeat("c", 64), DeploymentGeneration: view.Generation, SpecificationDigest: view.SpecificationDigest, CoreURL: fixturePublicURL} + if _, err = r.deployments.Enroll(t.Context(), token.Token, node); err != nil { + t.Fatal(err) + } + connection := uuid.NewString() + if err = r.deployments.ConnectNode(t.Context(), targetNode, connection, view.OwnerEpoch); err != nil { + t.Fatal(err) + } + compat := sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"} + if err = r.deployments.Heartbeat(t.Context(), targetNode, connection, view.OwnerEpoch, deployment.NodeHealth{ProviderReady: true}, []sandbox.GenerationStatus{{Generation: view.Generation, SpecificationDigest: view.SpecificationDigest, State: "ready", Checkpoint: &compat}}); err != nil { + t.Fatal(err) + } + state := runtimeCompute{Current: sandbox.Compute{ID: "source", Name: "source", Generation: 1}, Snapshot: &sandbox.SnapshotIdentity{ID: "snapshot", Compatibility: compat}} + raw, _ := json.Marshal(state) + if _, err = f.pool.Exec(t.Context(), `UPDATE environments SET initialization='complete',status='disconnected' WHERE id=$1`, owner.EnvironmentID); err != nil { + t.Fatal(err) + } + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspended',compute_state=$2,compute_revision=4,compute_retained_until=clock_timestamp()+interval '1 hour' WHERE id=$1`, owner.ID, raw); err != nil { + t.Fatal(err) + } + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET last_seen_at=clock_timestamp()-interval '1 minute' WHERE id=$1`, owner.NodeID); err != nil { + t.Fatal(err) + } + if operation == "wake" { + err = r.deployments.TouchActivity(t.Context(), owner.TenantID, owner.EnvironmentID) + } else { + _, err = f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID) + } + if err != nil { + t.Fatal(err) + } + source, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + if err = r.observe(t.Context(), source); err != nil { + t.Fatal(err) + } + moved, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + if moved.NodeID != targetNode || moved.ID != owner.ID || moved.DeviceID != owner.DeviceID || provider.derived != 0 || len(provider.requests) != 0 || provider.kills != 0 || provider.snapshots != 0 { + t.Fatal("source lane executed target effects or lost identity", moved) + } + // Simulate the independently scheduled target lane after the durable handoff. + r.nodeID = targetNode + err = r.observe(t.Context(), moved) + if operation == "wake" { + if !errors.Is(err, stop) || provider.derived != 1 || len(provider.requests) != 1 || provider.requests[0].ObserveOnly { + t.Fatal("target failed to use committed unsent restore", err) + } + } else { + if err != nil { + t.Fatal(err) + } + released, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || released.State != "released" || provider.kills != 0 || provider.snapshots != 1 { + t.Fatal("offline source cleanup did not settle at archive domain", released, err) + } + } + }) + } +} + +// Archive bytes may be unusable while the exact durable ownership receipt is +// still valid for deletion. Observing it must not authorize a new capture. +type cleanupReceiptProvider struct { + retentionProvider + observed bool +} + +func (p *cleanupReceiptProvider) Suspend(_ context.Context, q sandbox.SuspendRequest) (sandbox.ComputeState, error) { + p.observed = q.ObserveOnly + if !q.ObserveOnly { + return sandbox.ComputeState{}, sandbox.ErrComputeUnconfirmed + } + return sandbox.ComputeState{Compute: q.Source, Status: "unknown", Snapshot: &sandbox.SnapshotIdentity{ID: "owned-corrupt-archive", OperationID: q.OperationID, Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}}}, nil +} +func TestLostCaptureReceiptCanStillBeDeleted(t *testing.T) { + p := &cleanupReceiptProvider{} + f := workspaceSettlementFixtureWithSetup(t, p, workspaceNodeSetup(t, true)) + r, s := f.lifecycle, f.session + owner, err := r.provision(t.Context(), s.TenantID, s.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + state := runtimeCompute{Current: sandbox.Compute{ID: "source", Name: "source"}, SuspendID: uuid.NewString()} + raw, _ := json.Marshal(state) + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspending',compute_state=$2,compute_revision=4,compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID, raw); err != nil { + t.Fatal(err) + } + owner, err = r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + if err = r.observe(t.Context(), owner); err != nil { + t.Fatal(err) + } + current, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || current.State != "released" || !p.observed || p.kills != 1 || p.snapshots != 1 { + t.Fatal("exact cleanup receipt was discarded", current, err, p.observed, p.kills, p.snapshots) + } +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/allocations.go b/services/core/internal/persistence/postgres/deploymentpg/allocations.go index 8572502c5..7d67fff4b 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/allocations.go +++ b/services/core/internal/persistence/postgres/deploymentpg/allocations.go @@ -334,12 +334,31 @@ func (t *allocationTx) LoadActivity(current deployment.Allocation) (deployment.A return loadActivity(t.ctx, t.q, id) } -func (t *allocationTx) LoadRestore(current deployment.Allocation) (placement.Restore, error) { - node, err := parseID(current.NodeID) +func (t *allocationTx) LoadCheckpointPlacement(current deployment.Allocation) (placement.Deployment, []placement.CheckpointNode, error) { + d, err := placementpg.LockDeployment(t.ctx, t.q) if err != nil { - return placement.Restore{}, err + return d, nil, err } - return placementpg.LoadRestore(t.ctx, t.q, node, current.DeploymentGeneration) + nodes, err := placementpg.LoadNodes(t.ctx, t.q) + if err != nil { + return d, nil, err + } + candidates, err := placementpg.LoadCheckpointNodes(t.ctx, t.q, current.DeploymentGeneration, nodes) + return d, candidates, err +} + +func (t *allocationTx) MoveSuspended(current deployment.Allocation, nodeID string, change deployment.ComputeChange) (deployment.Allocation, error) { + source, err := parseID(current.NodeID) + if err != nil { + return deployment.Allocation{}, err + } + destination, err := parseID(nodeID) + if err != nil { + return deployment.Allocation{}, err + } + return t.change(current, func(ctx context.Context, id pgtype.UUID) (sqlc.RuntimeAllocation, error) { + return t.q.MoveSuspendedRuntimeCompute(ctx, sqlc.MoveSuspendedRuntimeComputeParams{ID: id, Source: source, Destination: destination, Revision: current.ComputeRevision, Phase: change.Phase, State: change.State}) + }) } func (t *allocationTx) ObserveRunning(current deployment.Allocation) (deployment.Allocation, error) { diff --git a/services/core/internal/persistence/postgres/deploymentpg/checkpoint_transfer_test.go b/services/core/internal/persistence/postgres/deploymentpg/checkpoint_transfer_test.go new file mode 100644 index 000000000..919f2dac6 --- /dev/null +++ b/services/core/internal/persistence/postgres/deploymentpg/checkpoint_transfer_test.go @@ -0,0 +1,301 @@ +package deploymentpg_test + +import ( + "encoding/json" + "errors" + "github.com/google/uuid" + "sync" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" +) + +var checkpointClass = sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-archive", ExecutionClass: "fixture-machine"} + +func readyCheckpointNode(t *testing.T, f fixture, node deployment.Enrollment, view deployment.View, compatibility sandbox.CheckpointCompatibility) { + t.Helper() + connection := uuid.NewString() + epoch, err := f.adapter.OwnerEpoch(t.Context()) + if err != nil { + t.Fatal(err) + } + if err := f.service.ConnectNode(t.Context(), node.NodeID, connection, epoch); err != nil { + t.Fatal(err) + } + if err := f.service.HeartbeatGenerations(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{}, []sandbox.GenerationStatus{{Generation: view.Generation, SpecificationDigest: view.SpecificationDigest, State: "ready", Checkpoint: &compatibility}}); err != nil { + t.Fatal(err) + } +} + +func suspendedCheckpointOwner(t *testing.T, f fixture, changes *deployment.ExecutionOperations, installation string, node deployment.Enrollment, view deployment.View) deployment.Allocation { + t.Helper() + owner := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspended',compute_state='{"snapshot":{"ID":"snapshot","Compatibility":{"artifact_domain":"fixture-archive","execution_class":"fixture-machine"}}}',compute_retained_until=clock_timestamp()+interval '24 hours',compute_wake_requested=true WHERE id=$1`, owner.ID); err != nil { + t.Fatal(err) + } + owner, err := f.adapter.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + return owner +} + +func TestCheckpointTransferPreservesAllocationAndFencesOldOwner(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, source, view, checkpointClass) + target := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, target, view, checkpointClass) + owner := suspendedCheckpointOwner(t, f, changes, installation, source, view) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + intent := json.RawMessage(`{"snapshot":{"ID":"snapshot"},"restore_id":"attempt"}`) + moved, err := changes.BeginRestore(t.Context(), owner, checkpointClass, intent) + if err != nil { + t.Fatal(err) + } + if moved.ID != owner.ID || moved.DeviceID != owner.DeviceID || moved.DeploymentGeneration != owner.DeploymentGeneration || moved.NodeID != target.NodeID || moved.ComputePhase != "restoring" || moved.ComputeRevision != owner.ComputeRevision+1 || !moved.ComputeRetainedUntil.Equal(*owner.ComputeRetainedUntil) { + t.Fatal("lost restore ownership", moved) + } + var placementNode, credentialNode string + if err := f.pool.QueryRow(t.Context(), `SELECT p.node_id::text,a.node_id::text FROM runtime_placements p JOIN runtime_allocations a ON a.environment_id=p.environment_id JOIN devices d ON d.id=a.device_id WHERE a.id=$1 AND d.revoked_at IS NULL`, owner.ID).Scan(&placementNode, &credentialNode); err != nil || placementNode != target.NodeID || credentialNode != target.NodeID { + t.Fatal("placement/device routing split", placementNode, credentialNode, err) + } + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil { + t.Fatal(err) + } + for _, n := range nodes { + if n.ID == source.NodeID && (n.Active != 0 || n.Retained != 0) || n.ID == target.NodeID && (n.Active != 1 || n.Retained != 1) { + t.Fatal("incorrect capacity", n) + } + } + if _, err := changes.BeginRestore(t.Context(), owner, checkpointClass, intent); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("stale source moved again", err) + } + if _, err := changes.BeginRestore(t.Context(), moved, checkpointClass, intent); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("unknown restore moved again", err) + } + if _, err := changes.RelocateCleanup(t.Context(), moved, checkpointClass); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("unknown restore reassigned cleanup", err) + } + +} + +func TestCheckpointTransferRequiresCompatibleGenerationAndCapacity(t *testing.T) { + for _, obstruction := range []string{"archive", "execution", "generation", "connection", "offline", "retained full", "active full", "expired", "quiescing", "no demand", "reset", "deleted", "epoch"} { + t.Run(obstruction, func(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + readyCheckpointNode(t, f, source, view, checkpointClass) + target := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, target, view, checkpointClass) + owner := suspendedCheckpointOwner(t, f, changes, installation, source, view) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + var statement string + arg := target.NodeID + want := placement.ErrNodeUnavailable + switch obstruction { + case "archive": + statement = `UPDATE runtime_node_generation_status SET checkpoint='{"artifact_domain":"other","execution_class":"fixture-machine"}' WHERE node_id=$1` + case "execution": + statement = `UPDATE runtime_node_generation_status SET checkpoint='{"artifact_domain":"fixture-archive","execution_class":"other"}' WHERE node_id=$1` + case "generation": + statement = `UPDATE runtime_node_generation_status SET generation=generation+1 WHERE node_id=$1` + case "connection": + statement = `UPDATE runtime_node_generation_status SET connection_id=gen_random_uuid() WHERE node_id=$1` + case "offline": + statement = `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1` + case "retained full": + suspendedCheckpointOwner(t, f, changes, installation, target, view) + case "active full": + runningPressureOwner(t, f, changes, installation, target.NodeID, view.Generation) + case "expired": + statement = `UPDATE runtime_allocations SET compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1` + arg = owner.ID + want = deployment.ErrAllocationConflict + case "quiescing": + statement = `UPDATE runtime_allocations SET compute_phase='quiescing' WHERE id=$1` + arg = owner.ID + want = deployment.ErrAllocationConflict + case "no demand": + statement = `UPDATE runtime_allocations SET compute_wake_requested=false WHERE id=$1` + arg = owner.ID + want = deployment.ErrAllocationConflict + case "deleted": + statement = `UPDATE sessions SET deleted_at=clock_timestamp() WHERE id=(SELECT session_id FROM environments WHERE id=$1)` + arg = owner.EnvironmentID + want = sessions.ErrNotFound + case "epoch": + statement = `UPDATE runtime_node_generation_status SET owner_epoch=owner_epoch+1 WHERE node_id=$1` + case "reset": + statement = `UPDATE runtime_deployment SET reset_clear='force',reset_requested_at=clock_timestamp(),reset_forced_at=clock_timestamp(),reset_audit='{}' WHERE installation_id=$1` + arg = installation + want = placement.ErrResetAdmission + } + if statement != "" { + if _, err := f.pool.Exec(t.Context(), statement, arg); err != nil { + t.Fatal(err) + } + } + if _, err := changes.BeginRestore(t.Context(), owner, checkpointClass, json.RawMessage(`{"restore_id":"attempt"}`)); !errors.Is(err, want) { + t.Fatal("invalid destination admitted", err, want) + } + current, err := f.adapter.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || current.NodeID != source.NodeID || current.ComputeRevision != owner.ComputeRevision { + t.Fatal("rejection changed ownership", current, err) + } + }) + } +} + +func TestCheckpointTransfersSerializeDestinationCapacity(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 2, MaxRetained: 2}) + readyCheckpointNode(t, f, source, view, checkpointClass) + target := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, target, view, checkpointClass) + owners := []deployment.Allocation{suspendedCheckpointOwner(t, f, changes, installation, source, view), suspendedCheckpointOwner(t, f, changes, installation, source, view)} + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + errs := make([]error, len(owners)) + var wg sync.WaitGroup + for i, owner := range owners { + wg.Go(func() { + _, errs[i] = changes.BeginRestore(t.Context(), owner, checkpointClass, json.RawMessage(`{"restore_id":"attempt"}`)) + }) + } + wg.Wait() + success := 0 + for _, err := range errs { + if err == nil { + success++ + } else if !errors.Is(err, placement.ErrNodeUnavailable) { + t.Fatal(err) + } + } + if success != 1 { + t.Fatal("destination overbooked", errs) + } +} + +func TestCheckpointCleanupRelocatesArtifactWithoutExecutionCompatibility(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, source, view, checkpointClass) + target := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + other := checkpointClass + other.ExecutionClass = "other-machine" + readyCheckpointNode(t, f, target, view, other) + runningPressureOwner(t, f, changes, installation, target.NodeID, view.Generation) + owner := suspendedCheckpointOwner(t, f, changes, installation, source, view) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + if _, err := changes.RelocateCleanup(t.Context(), owner, checkpointClass); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("moved live allocation for cleanup", err) + } + cleanup, err := changes.RequestCleanup(t.Context(), owner) + if err != nil { + t.Fatal(err) + } + moved, err := changes.RelocateCleanup(t.Context(), cleanup, checkpointClass) + if err != nil { + t.Fatal(err) + } + if moved.NodeID != target.NodeID || moved.State != "cleanup_pending" || moved.ComputePhase != "suspended" || moved.ID != owner.ID || !moved.ComputeRetainedUntil.Equal(*owner.ComputeRetainedUntil) { + t.Fatal("cleanup recreated compute", moved) + } + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil { + t.Fatal(err) + } + for _, n := range nodes { + if n.ID == target.NodeID && (n.Active != 1 || n.Retained != 2) { + t.Fatal("cleanup released retained slot", n) + } + } + if _, err := changes.ReleaseAllocation(t.Context(), moved); err != nil { + t.Fatal(err) + } +} + +func TestCheckpointPressureSuspendsOnlyUsableDestination(t *testing.T) { + for _, scenario := range []string{"waiting restore", "retained full", "incompatible", "other free"} { + t.Run(scenario, func(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, source, view, checkpointClass) + suspendedCheckpointOwner(t, f, changes, installation, source, view) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + retained := 2 + if scenario == "retained full" { + retained = 1 + } + target := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: retained}) + compatibility := checkpointClass + if scenario == "incompatible" { + compatibility.ExecutionClass = "other" + } + readyCheckpointNode(t, f, target, view, compatibility) + active := runningPressureOwner(t, f, changes, installation, target.NodeID, view.Generation) + if scenario == "other free" { + other := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, other, view, checkpointClass) + } + until := time.Now().Add(24 * time.Hour) + _, err := changes.SetCompute(t.Context(), active, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute) + if scenario == "waiting restore" { + if err != nil { + t.Fatal("compatible wake cannot trigger pressure suspension", err) + } + } else if !errors.Is(err, deployment.ErrNotIdle) { + t.Fatal("suspended without usable restore capacity", err) + } + }) + } +} + +func TestCheckpointRestoreUsesRetainedSlotAndImmutableGeneration(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, source, view, checkpointClass) + owner := suspendedCheckpointOwner(t, f, changes, installation, source, view) + // A rollout must not substitute the serving generation for this snapshot. + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET ready_generation=ready_generation+1 WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + restored, err := changes.BeginRestore(t.Context(), owner, checkpointClass, json.RawMessage(`{"restore_id":"attempt"}`)) + if err != nil { + t.Fatal(err) + } + if restored.NodeID != source.NodeID || restored.DeploymentGeneration != owner.DeploymentGeneration { + t.Fatal("changed retained generation", restored) + } + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil || len(nodes) != 1 || nodes[0].Active != 1 || nodes[0].Retained != 1 { + t.Fatal("restore reserved a second retained slot", nodes, err) + } +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/fixture_test.go b/services/core/internal/persistence/postgres/deploymentpg/fixture_test.go index 8c49f92f0..8032f8288 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/fixture_test.go +++ b/services/core/internal/persistence/postgres/deploymentpg/fixture_test.go @@ -3,22 +3,21 @@ package deploymentpg_test import ( "bytes" "context" - "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "strings" "testing" - "github.com/google/uuid" - "github.com/jackc/pgx/v5/pgxpool" - "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/adminaudit" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/credentialcrypto" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/deploymentpg" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/pgtest" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/pgunit" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/providers" + "github.com/google/uuid" + "github.com/jackc/pgx/v5/pgxpool" ) const fixturePublicURL = "https://core.example" @@ -146,8 +145,22 @@ func (f fixture) connect(t *testing.T, nodeID string) string { if err := f.service.ConnectNode(t.Context(), nodeID, connection, epoch); err != nil { t.Fatal(err) } - if err := f.service.Heartbeat(t.Context(), nodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := f.heartbeat(t.Context(), nodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}, nil); err != nil { t.Fatal(err) } return connection } + +// heartbeat reports the fixture's verified generation alongside static readiness. +func (f fixture) heartbeat(ctx context.Context, nodeID, connection string, epoch uint64, health deployment.NodeHealth, _ []sandbox.GenerationStatus) error { + var generation uint64 + var digest, provider string + if err := f.pool.QueryRow(ctx, "SELECT n.deployment_generation, n.specification_digest, d.provider_kind FROM runtime_nodes n CROSS JOIN runtime_deployment d WHERE n.id=$1", nodeID).Scan(&generation, &digest, &provider); err != nil { + return err + } + var statuses []sandbox.GenerationStatus + if health.ProviderReady && provider == "microsandbox" { + statuses = []sandbox.GenerationStatus{{Generation: generation, SpecificationDigest: digest, State: "ready", Checkpoint: &sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}}} + } + return f.service.Heartbeat(ctx, nodeID, connection, epoch, health, statuses) +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/host_history_test.go b/services/core/internal/persistence/postgres/deploymentpg/host_history_test.go index ee83493e4..4b7c09e2a 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/host_history_test.go +++ b/services/core/internal/persistence/postgres/deploymentpg/host_history_test.go @@ -36,7 +36,7 @@ func TestNodeHostHistorySamplingAndDetail(t *testing.T) { now := time.Now().UTC() host := &deployment.NodeHost{EffectiveCPUCores: hostHistoryPtr(2.0), CPUUtilization: hostHistoryPtr(0.35), TotalMemoryBytes: hostHistoryPtr(int64(4096)), AvailableMemoryBytes: hostHistoryPtr(int64(1024)), AvailableDiskBytes: hostHistoryPtr(int64(8192)), ObservedAt: &now} health := deployment.NodeHealth{ProviderReady: true, Host: host} - if err := f.service.Heartbeat(t.Context(), node, connection, epoch, health); err != nil { + if err := f.heartbeat(t.Context(), node, connection, epoch, health, nil); err != nil { t.Fatal(err) } for i, want := range []int64{1, 0} { @@ -113,7 +113,7 @@ func TestNodeHostHistorySamplingAndDetail(t *testing.T) { // A disconnected node cannot create another history row, even with a fresh last observation. next := now.Add(time.Millisecond) host.ObservedAt = &next - if err := f.service.Heartbeat(t.Context(), node, connection, epoch, health); err != nil { + if err := f.heartbeat(t.Context(), node, connection, epoch, health, nil); err != nil { t.Fatal(err) } if err := f.service.DisconnectNode(t.Context(), node, connection, epoch); err != nil { @@ -136,7 +136,7 @@ func TestNodeHostHistorySamplingAndDetail(t *testing.T) { func TestNodeHostHistoryFencingAndUnknown(t *testing.T) { f, node, conn, epoch := hostHistoryNode(t) for _, at := range []time.Time{time.Now().Add(-time.Minute), time.Now().Add(time.Hour)} { - if err := f.service.Heartbeat(t.Context(), node, conn, epoch, deployment.NodeHealth{Host: &deployment.NodeHost{ObservedAt: &at}}); err != nil { + if err := f.heartbeat(t.Context(), node, conn, epoch, deployment.NodeHealth{Host: &deployment.NodeHost{ObservedAt: &at}}, nil); err != nil { t.Fatal(err) } if n, err := f.adapter.SampleHostHistory(t.Context()); err != nil || n != 0 { @@ -145,10 +145,10 @@ func TestNodeHostHistoryFencingAndUnknown(t *testing.T) { } now := time.Now().UTC() health := deployment.NodeHealth{Host: &deployment.NodeHost{ObservedAt: &now}} - if err := f.service.Heartbeat(t.Context(), node, uuid.NewString(), epoch, health); !errors.Is(err, deployment.ErrNodeCredential) { + if err := f.heartbeat(t.Context(), node, uuid.NewString(), epoch, health, nil); !errors.Is(err, deployment.ErrNodeCredential) { t.Fatal(err) } - if err := f.service.Heartbeat(t.Context(), node, conn, epoch, health); err != nil { + if err := f.heartbeat(t.Context(), node, conn, epoch, health, nil); err != nil { t.Fatal(err) } if n, err := f.adapter.SampleHostHistory(t.Context()); err != nil || n != 1 { @@ -164,7 +164,7 @@ func TestNodeHostHistoryFencingAndUnknown(t *testing.T) { {ObservedAt: &now, TotalMemoryBytes: hostHistoryPtr(int64(10)), AvailableMemoryBytes: hostHistoryPtr(int64(11))}, {ObservedAt: &now, AvailableDiskBytes: hostHistoryPtr(int64(-1))}, {ObservedAt: &now, EffectiveCPUCores: hostHistoryPtr(0.0)}, {}, } { - if err := f.service.Heartbeat(t.Context(), node, conn, epoch, deployment.NodeHealth{Host: host}); !errors.Is(err, deployment.ErrInvalidInput) { + if err := f.heartbeat(t.Context(), node, conn, epoch, deployment.NodeHealth{Host: host}, nil); !errors.Is(err, deployment.ErrInvalidInput) { t.Fatal(host, err) } } diff --git a/services/core/internal/persistence/postgres/deploymentpg/presence_test.go b/services/core/internal/persistence/postgres/deploymentpg/presence_test.go index 52b788997..83957fb31 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/presence_test.go +++ b/services/core/internal/persistence/postgres/deploymentpg/presence_test.go @@ -246,7 +246,7 @@ func TestNodeStaleEpochCannotReplaceCurrentConnection(t *testing.T) { if err := f.service.DisconnectNode(t.Context(), node.NodeID, connection, epoch); err != nil { t.Fatal(err) } - if err := f.service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: false}); !errors.Is(err, deployment.ErrNodeCredential) { + if err := f.heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: false}, nil); !errors.Is(err, deployment.ErrNodeCredential) { t.Fatal("old Core rewrote health", err) } nodes, err := f.service.ListNodes(t.Context()) @@ -277,7 +277,7 @@ func TestNodeDiagnosticReachesListAndDetail(t *testing.T) { {reported: "dial unix /var/run/docker.sock: permission denied", want: "provider_unavailable"}, {reported: "artifacts_unavailable", want: "", ready: true}, } { - if err := f.service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: tc.ready, Diagnostic: sandbox.NodeDiagnosticCode(tc.reported)}); err != nil { + if err := f.heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: tc.ready, Diagnostic: sandbox.NodeDiagnosticCode(tc.reported)}, nil); err != nil { t.Fatal(tc.reported, err) } list, err := f.service.ListNodes(t.Context()) @@ -321,11 +321,11 @@ func TestNodeStatusUsesAuthenticatedFreshPresence(t *testing.T) { if err != nil { t.Fatal(err) } - if err := f.service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: false}); err != nil { + if err := f.heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: false}, nil); err != nil { t.Fatal(err) } status(true, false) - if err := f.service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := f.heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}, nil); err != nil { t.Fatal(err) } status(true, true) diff --git a/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure.go b/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure.go index 1f51fb9f8..aa1d7f0fd 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure.go +++ b/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure.go @@ -2,12 +2,15 @@ package deploymentpg import ( "context" + "encoding/json" "github.com/jackc/pgx/v5/pgtype" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/db/sqlc" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/placementpg" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" ) func (t *allocationTx) LoadGenerationSpecification(generation uint64) (deployment.GenerationSpecification, error) { @@ -32,23 +35,28 @@ func (t *allocationTx) LoadSuspensionDemand(current deployment.Allocation) (depl if err != nil { return result, err } - node, err := parseID(current.NodeID) + restores, err := t.q.ListWaitingCheckpointRestores(t.ctx) if err != nil { return result, err } - generations, err := t.q.ListWaitingRuntimeRestoreGenerations(t.ctx, node) - if err != nil { - return result, err - } - for _, generation := range generations { - ready, err := t.q.NodeGenerationReady(t.ctx, sqlc.NodeGenerationReadyParams{NodeID: node, Generation: generation.Int64}) - if err != nil { + byGeneration := map[int64][]placement.CheckpointNode{} + for _, restore := range restores { + var compatibility sandbox.CheckpointCompatibility + if err := json.Unmarshal(restore.Checkpoint, &compatibility); err != nil { return result, err } - if ready { - result.RestoreWaiting = true - return result, nil + if compatibility.Validate() != nil { + continue + } + nodes, ok := byGeneration[restore.DeploymentGeneration.Int64] + if !ok { + nodes, err = placementpg.LoadCheckpointNodes(t.ctx, t.q, uint64(restore.DeploymentGeneration.Int64), result.Nodes) + if err != nil { + return result, err + } + byGeneration[restore.DeploymentGeneration.Int64] = nodes } + result.CheckpointRestores = append(result.CheckpointRestores, deployment.CheckpointRestoreDemand{SourceNode: uuidString(restore.NodeID), Compatibility: compatibility, Nodes: nodes}) } return result, nil } diff --git a/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure_test.go b/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure_test.go index 4d3b419e0..77fed6a34 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure_test.go +++ b/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure_test.go @@ -132,13 +132,16 @@ func TestPressureSuspensionServesRestoreOnItsOriginalNode(t *testing.T) { node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) f.connect(t, node.NodeID) waiting := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) - if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspended',compute_retained_until=clock_timestamp()+interval '1 hour',compute_wake_requested=true WHERE id=$1`, waiting.ID); err != nil { + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspended',compute_state='{"snapshot":{"Compatibility":{"artifact_domain":"fixture-store","execution_class":"fixture-runtime"}}}',compute_retained_until=clock_timestamp()+interval '1 hour',compute_wake_requested=true WHERE id=$1`, waiting.ID); err != nil { t.Fatal(err) } owner := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) - // Free capacity on another node cannot restore this retained compute. + // Free capacity in another artifact domain cannot restore this checkpoint. other := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) f.connect(t, other.NodeID) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_node_generation_status SET checkpoint='{"artifact_domain":"other","execution_class":"fixture-runtime"}' WHERE node_id=$1`, other.NodeID); err != nil { + t.Fatal(err) + } until := time.Now().Add(time.Hour) if _, err := changes.SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute); err != nil { t.Fatal(err) diff --git a/services/core/internal/persistence/postgres/deploymentpg/tx.go b/services/core/internal/persistence/postgres/deploymentpg/tx.go index 28f35e1c5..4441dd7d0 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/tx.go +++ b/services/core/internal/persistence/postgres/deploymentpg/tx.go @@ -237,7 +237,14 @@ func (t *nodeTx) UpsertGenerationStatus(status deployment.GenerationStatusRecord if err != nil { return err } - return t.q.UpsertNodeGenerationStatus(t.ctx, sqlc.UpsertNodeGenerationStatusParams{NodeID: id, Generation: int64(status.Generation), SpecificationDigest: status.SpecificationDigest, ConnectionID: connection, OwnerEpoch: int64(status.OwnerEpoch), State: status.State, Diagnostic: status.Diagnostic}) + var checkpoint []byte + if status.Checkpoint != nil { + checkpoint, err = json.Marshal(status.Checkpoint) + if err != nil { + return err + } + } + return t.q.UpsertNodeGenerationStatus(t.ctx, sqlc.UpsertNodeGenerationStatusParams{NodeID: id, Generation: int64(status.Generation), SpecificationDigest: status.SpecificationDigest, ConnectionID: connection, OwnerEpoch: int64(status.OwnerEpoch), State: status.State, Diagnostic: status.Diagnostic, Checkpoint: checkpoint}) } func (t *nodeTx) PromoteServingGeneration(nodeID string, generation uint64) error { diff --git a/services/core/internal/persistence/postgres/placementpg/placementpg.go b/services/core/internal/persistence/postgres/placementpg/placementpg.go index 73fc46b35..ad5abde8a 100644 --- a/services/core/internal/persistence/postgres/placementpg/placementpg.go +++ b/services/core/internal/persistence/postgres/placementpg/placementpg.go @@ -7,6 +7,7 @@ package placementpg import ( "context" + "encoding/json" "github.com/google/uuid" "github.com/jackc/pgx/v5/pgtype" @@ -14,6 +15,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/db/sqlc" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/pgunit" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" ) // LockDeployment locks the deployment and returns it as placement reads it. @@ -65,21 +67,34 @@ func ReleasePlacement(ctx context.Context, q *sqlc.Queries, environment pgtype.U return q.ReleaseRuntimePlacement(ctx, environment) } -// LoadRestore locks the deployment and returns what restoring suspended -// compute of the generation on the node reads. -func LoadRestore(ctx context.Context, q *sqlc.Queries, nodeID pgtype.UUID, generation uint64) (placement.Restore, error) { - if _, err := q.LockRuntimeDeployment(ctx); err != nil { - return placement.Restore{}, err +// LoadCheckpointNodes joins current presence/capacity with exact-generation +// readiness. The caller holds the deployment lock for both observations. +func LoadCheckpointNodes(ctx context.Context, q *sqlc.Queries, generation uint64, nodes []placement.Node) ([]placement.CheckpointNode, error) { + ready, err := q.ListCheckpointGenerationNodes(ctx, int64(generation)) + if err != nil { + return nil, err + } + byID := make(map[string]placement.Node, len(nodes)) + for _, node := range nodes { + byID[node.ID] = node } - restore := placement.Restore{Generation: generation} - rows, err := q.ListRuntimeNodes(ctx, nodeID) - if err != nil || len(rows) == 0 { - return restore, err + candidates := make([]placement.CheckpointNode, 0, len(ready)) + for _, row := range ready { + if len(row.Checkpoint) == 0 { + continue + } + var compatibility sandbox.CheckpointCompatibility + if err := json.Unmarshal(row.Checkpoint, &compatibility); err != nil { + return nil, err + } + if err := compatibility.Validate(); err != nil { + return nil, err + } + if node, ok := byID[uuidString(row.ID)]; ok { + candidates = append(candidates, placement.CheckpointNode{Node: node, Checkpoint: &compatibility}) + } } - n := node(rows[0]) - restore.Node = &n - restore.GenerationReady, err = q.NodeGenerationReady(ctx, sqlc.NodeGenerationReadyParams{NodeID: nodeID, Generation: int64(generation)}) - return restore, err + return candidates, nil } // ComputeBlocksAdmission reports whether the Session's managed compute is in diff --git a/services/core/internal/sandbox/docker/node.go b/services/core/internal/sandbox/docker/node.go index 115d6bf0c..b8ddc04fb 100644 --- a/services/core/internal/sandbox/docker/node.go +++ b/services/core/internal/sandbox/docker/node.go @@ -68,7 +68,9 @@ func BuildNode(config sandbox.NodeConfig, _ sandbox.LocalOptions, result *sandbo return func() {}, errors.New("invalid managed Docker provider configuration") } result.Provider = provider - result.Probe = dockerProbe(c, entry.Image, config.Specification.Resources) + result.Probe = func(ctx context.Context) (*sandbox.CheckpointCompatibility, error) { + return nil, dockerProbe(c, entry.Image, config.Specification.Resources)(ctx) + } result.BackendFingerprint = sandbox.BackendFingerprint(config.Provider, entry.Host) return closeProvider, nil } diff --git a/services/core/internal/sandbox/generation.go b/services/core/internal/sandbox/generation.go index f191d2859..66bbbda8d 100644 --- a/services/core/internal/sandbox/generation.go +++ b/services/core/internal/sandbox/generation.go @@ -10,10 +10,11 @@ import ( // GenerationStatus is one sparse observation, not a complete provider inventory. // A serving pin is owned by Core and is independent of target preparation. type GenerationStatus struct { - Generation uint64 `json:"generation"` - SpecificationDigest string `json:"specification_digest"` - State string `json:"state"` - Diagnostic string `json:"diagnostic,omitempty"` + Generation uint64 `json:"generation"` + SpecificationDigest string `json:"specification_digest"` + State string `json:"state"` + Diagnostic string `json:"diagnostic,omitempty"` + Checkpoint *CheckpointCompatibility `json:"checkpoint,omitempty"` } type GenerationReference struct { diff --git a/services/core/internal/sandbox/microsandbox/checkpoint_linux.go b/services/core/internal/sandbox/microsandbox/checkpoint_linux.go new file mode 100644 index 000000000..e28018709 --- /dev/null +++ b/services/core/internal/sandbox/microsandbox/checkpoint_linux.go @@ -0,0 +1,133 @@ +//go:build linux + +package microsandbox + +import ( + "crypto/sha256" + "encoding/hex" + "encoding/json" + "io" + "os" + "path/filepath" + "sort" + "strings" + "syscall" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +// CheckpointStore opens only an explicitly configured, private archive root. +// Its marker is an operator-owned identity, never inferred from workspace paths. +func CheckpointStore(c Config) (string, error) { + if !filepath.IsAbs(c.CheckpointRoot) || filepath.Clean(c.CheckpointRoot) != c.CheckpointRoot { + return "", sandbox.ErrInvalid + } + real, err := filepath.EvalSymlinks(c.CheckpointRoot) + if err != nil { + return "", err + } + if real != c.CheckpointRoot { + return "", sandbox.ErrOwnership + } + fd, err := syscall.Open(c.CheckpointRoot, syscall.O_RDONLY|syscall.O_DIRECTORY|syscall.O_NOFOLLOW|syscall.O_CLOEXEC, 0) + if err != nil { + return "", err + } + f := os.NewFile(uintptr(fd), c.CheckpointRoot) + defer f.Close() + st, err := f.Stat() + if err != nil { + return "", err + } + if st.Mode().Perm() != 0700 || st.Sys().(*syscall.Stat_t).Uid != uint32(os.Geteuid()) { + return "", sandbox.ErrOwnership + } + marker, err := syscall.Openat(fd, ".oac-checkpoint-store", syscall.O_RDONLY|syscall.O_NOFOLLOW|syscall.O_CLOEXEC, 0) + if err != nil { + return "", err + } + m := os.NewFile(uintptr(marker), "checkpoint marker") + defer m.Close() + st, err = m.Stat() + if err != nil { + return "", err + } + if !st.Mode().IsRegular() || st.Mode().Perm() != 0600 || st.Sys().(*syscall.Stat_t).Uid != uint32(os.Geteuid()) { + return "", sandbox.ErrOwnership + } + data, err := io.ReadAll(io.LimitReader(m, 38)) + if err != nil { + return "", err + } + id := strings.TrimSuffix(string(data), "\n") + if !validID(id) || string(data) != id+"\n" { + return "", sandbox.ErrOwnership + } + h := sha256.Sum256([]byte(c.InstallationID + ":" + id)) + return hex.EncodeToString(h[:]), nil +} + +// CheckpointClass deliberately admits only identical native and host execution +// profiles. Paths, hostname and local SDK cache identities are not compatibility. +func CheckpointClass(c Config) (sandbox.CheckpointCompatibility, error) { + domain, err := CheckpointStore(c) + if err != nil { + return sandbox.CheckpointCompatibility{}, err + } + cpu, err := os.ReadFile("/proc/cpuinfo") + if err != nil { + return sandbox.CheckpointCompatibility{}, err + } + stable := checkpointCPUProfiles(string(cpu)) + if len(stable) == 0 { + return sandbox.CheckpointCompatibility{}, sandbox.ErrInvalid + } + kernel, err := os.ReadFile("/proc/sys/kernel/osrelease") + if err != nil { + return sandbox.CheckpointCompatibility{}, err + } + raw, _ := json.Marshal(struct { + SDK, Runtime, Firmware, Image, Kernel string + CPUs uint8 + Memory, Root, Environment uint32 + CPU []string + }{SDKVersion, c.RuntimeSHA256, c.FirmwareSHA256, c.Image, strings.TrimSpace(string(kernel)), c.CPUs, c.MemoryMiB, c.RootDiskMiB, c.EnvironmentDiskMiB, stable}) + h := sha256.Sum256(raw) + return sandbox.CheckpointCompatibility{ArtifactDomain: domain, ExecutionClass: hex.EncodeToString(h[:])}, nil +} + +// Preserve each processor's feature/model association. Multiplicity and logical +// processor indexes are irrelevant; distinct heterogeneous profiles are not. +func checkpointCPUProfiles(cpuinfo string) []string { + profiles := map[string]bool{} + for _, block := range strings.Split(strings.TrimSpace(cpuinfo), "\n\n") { + fields := []string{} + for _, line := range strings.Split(block, "\n") { + key, value, ok := strings.Cut(line, ":") + if !ok { + continue + } + key = strings.TrimSpace(key) + switch key { + case "vendor_id", "cpu family", "model", "stepping", "flags", "Features", "CPU implementer", "CPU architecture", "CPU part", "CPU revision": + value = strings.TrimSpace(value) + if key == "flags" || key == "Features" { + features := strings.Fields(value) + sort.Strings(features) + value = strings.Join(features, " ") + } + fields = append(fields, key+":"+value) + } + } + if len(fields) > 0 { + sort.Strings(fields) + profiles[strings.Join(fields, "\n")] = true + } + } + result := make([]string, 0, len(profiles)) + for profile := range profiles { + result = append(result, profile) + } + sort.Strings(result) + return result +} diff --git a/services/core/internal/sandbox/microsandbox/checkpoint_linux_test.go b/services/core/internal/sandbox/microsandbox/checkpoint_linux_test.go new file mode 100644 index 000000000..0d296776a --- /dev/null +++ b/services/core/internal/sandbox/microsandbox/checkpoint_linux_test.go @@ -0,0 +1,88 @@ +//go:build linux + +package microsandbox + +import ( + "errors" + "os" + "path/filepath" + "reflect" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +func privateCheckpointRoot(t *testing.T, marker string) string { + t.Helper() + root := t.TempDir() + if err := os.Chmod(root, 0700); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(root, ".oac-checkpoint-store"), []byte(marker+"\n"), 0600); err != nil { + t.Fatal(err) + } + return root +} +func TestCheckpointCompatibilityUsesStoreIdentityNotMountOrCachePath(t *testing.T) { + c := testConfig() + c.CheckpointRoot = privateCheckpointRoot(t, "11111111-1111-4111-8111-111111111111") + a, err := CheckpointClass(c) + if err != nil { + t.Fatal(err) + } + c.CheckpointRoot = privateCheckpointRoot(t, "11111111-1111-4111-8111-111111111111") + c.RuntimeHome = "/different/cache" + c.RuntimePath = "/different/msb" + c.FirmwarePath = "/different/firmware" + b, err := CheckpointClass(c) + if err != nil || a != b { + t.Fatal("host paths changed compatibility", a, b, err) + } + c.MemoryMiB++ + b, err = CheckpointClass(c) + if err != nil || a.ExecutionClass == b.ExecutionClass || a.ArtifactDomain != b.ArtifactDomain { + t.Fatal("geometry not fenced", a, b, err) + } + c.CheckpointRoot = privateCheckpointRoot(t, "22222222-2222-4222-8222-222222222222") + b, err = CheckpointClass(c) + if err != nil || a.ArtifactDomain == b.ArtifactDomain { + t.Fatal("foreign store matched", a, b, err) + } +} +func TestCheckpointStoreRequiresPrivateCanonicalIdentity(t *testing.T) { + c := testConfig() + c.CheckpointRoot = privateCheckpointRoot(t, "11111111-1111-4111-8111-111111111111") + if err := os.Chmod(c.CheckpointRoot, 0755); err != nil { + t.Fatal(err) + } + if _, err := CheckpointStore(c); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal(err) + } + if err := os.Chmod(c.CheckpointRoot, 0700); err != nil { + t.Fatal(err) + } + link := filepath.Join(t.TempDir(), "link") + if err := os.Symlink(c.CheckpointRoot, link); err != nil { + t.Fatal(err) + } + c.CheckpointRoot = link + if _, err := CheckpointStore(c); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal(err) + } + c.CheckpointRoot = privateCheckpointRoot(t, "00000000-0000-0000-0000-000000000000") + if _, err := CheckpointStore(c); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal(err) + } +} + +func TestCPUCompatibilityPreservesHeterogeneousProcessorProfiles(t *testing.T) { + first := "processor:0\nmodel:1\nflags:a b\n\nprocessor:1\nmodel:2\nflags:c d\n" + reordered := "processor:9\nflags:d c\nmodel:2\n\nprocessor:4\nflags:b a\nmodel:1\n\nprocessor:5\nflags:a b\nmodel:1\n" + crossed := "processor:0\nmodel:1\nflags:c d\n\nprocessor:1\nmodel:2\nflags:a b\n" + if !reflect.DeepEqual(checkpointCPUProfiles(first), checkpointCPUProfiles(reordered)) { + t.Fatal("processor order/count affected class") + } + if reflect.DeepEqual(checkpointCPUProfiles(first), checkpointCPUProfiles(crossed)) { + t.Fatal("heterogeneous CPU associations collapsed") + } +} diff --git a/services/core/internal/sandbox/microsandbox/checkpoint_other.go b/services/core/internal/sandbox/microsandbox/checkpoint_other.go new file mode 100644 index 000000000..dc92ab8d7 --- /dev/null +++ b/services/core/internal/sandbox/microsandbox/checkpoint_other.go @@ -0,0 +1,10 @@ +//go:build !linux + +package microsandbox + +import "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + +func CheckpointStore(Config) (string, error) { return "", sandbox.ErrHostUnsupported } +func CheckpointClass(Config) (sandbox.CheckpointCompatibility, error) { + return sandbox.CheckpointCompatibility{}, sandbox.ErrHostUnsupported +} diff --git a/services/core/internal/sandbox/microsandbox/identity.go b/services/core/internal/sandbox/microsandbox/identity.go index b831c016d..80ddc2840 100644 --- a/services/core/internal/sandbox/microsandbox/identity.go +++ b/services/core/internal/sandbox/microsandbox/identity.go @@ -24,6 +24,9 @@ func validHash(s string) bool { return e == nil && len(b) == 32 && strings.ToLower(s) == s } func (c Config) Validate() error { + if !filepath.IsAbs(c.CheckpointRoot) || filepath.Clean(c.CheckpointRoot) != c.CheckpointRoot { + return sandbox.ErrInvalid + } if !validID(c.InstallationID) || !filepath.IsAbs(c.HelperPath) || !filepath.IsAbs(c.RuntimeHome) || !filepath.IsAbs(c.RuntimePath) || !filepath.IsAbs(c.FirmwarePath) || !validHash(c.RuntimeSHA256) || !validHash(c.FirmwareSHA256) || c.MemoryMiB == 0 || c.CPUs == 0 || c.RootDiskMiB == 0 || (!c.ExternalWorkspace && c.EnvironmentDiskMiB == 0) { return sandbox.ErrInvalid } @@ -70,7 +73,7 @@ func ValidateCompute(c Config, r sandbox.Reference, v Compute) error { return nil } func ValidateSnapshot(c Config, r sandbox.Reference, s SnapshotIdentity) error { - if !ValidReference(r) || !validID(s.OperationID) || s.Reference != SnapshotReference(c, r, s.OperationID) || s.SourceName != Name(c, r, s.SourceGeneration) || s.SourceID == "" || s.ID == "" || s.Digest == "" || s.CheckpointID == "" || s.CheckpointRoot == "" { + if s.Compatibility.Validate() != nil || !ValidReference(r) || !validID(s.OperationID) || s.Reference != SnapshotReference(c, r, s.OperationID) || s.SourceName != Name(c, r, s.SourceGeneration) || s.SourceID == "" || s.ID == "" || s.Digest == "" || s.CheckpointID == "" || s.CheckpointRoot == "" { return sandbox.ErrInvalid } return nil @@ -99,6 +102,12 @@ func ValidateRequest(q Request) error { if q.Workspace != nil && ((q.Operation != "create" && q.Operation != "resume") || !validID(q.Workspace.ObjectID) || !filepath.IsAbs(q.Workspace.Path) || filepath.Clean(q.Workspace.Path) != q.Workspace.Path || strings.ContainsRune(q.Workspace.Path, 0)) { return sandbox.ErrInvalid } + if q.Workspace != nil { + relative, err := filepath.Rel(q.Workspace.Path, q.Config.CheckpointRoot) + if err != nil || relative == "." || (relative != ".." && !strings.HasPrefix(relative, ".."+string(filepath.Separator))) { + return sandbox.ErrInvalid + } + } switch q.Operation { case "create": if q.Config.ExternalWorkspace != (q.Workspace != nil) { diff --git a/services/core/internal/sandbox/microsandbox/node.go b/services/core/internal/sandbox/microsandbox/node.go index 8f2789053..2b68552dc 100644 --- a/services/core/internal/sandbox/microsandbox/node.go +++ b/services/core/internal/sandbox/microsandbox/node.go @@ -1,6 +1,7 @@ package microsandbox import ( + "context" "errors" "os" "path/filepath" @@ -23,11 +24,12 @@ var NodeArtifacts = []providerassets.Artifact{ // The helper owns local paths; no ambient backend is selected. Resources, the // image and the artifact hashes come from the deployment specification. type Native struct { - HelperPath string `json:"helper_path"` - RuntimeHome string `json:"runtime_home"` - RuntimePath string `json:"runtime_path"` - FirmwarePath string `json:"firmware_path"` - Network Network `json:"network"` + HelperPath string `json:"helper_path"` + RuntimeHome string `json:"runtime_home"` + CheckpointRoot string `json:"checkpoint_root"` + RuntimePath string `json:"runtime_path"` + FirmwarePath string `json:"firmware_path"` + Network Network `json:"network"` } type Network struct { @@ -55,7 +57,7 @@ func configureMicrosandbox(entry Native, spec sandbox.DeploymentSpec, caller *Pr release, resources := spec.Runtime, spec.Resources config := Config{ ExternalWorkspace: spec.Workspace != nil, - InstallationID: result.InstallationID, HelperPath: entry.HelperPath, RuntimeHome: entry.RuntimeHome, RuntimePath: entry.RuntimePath, FirmwarePath: entry.FirmwarePath, + InstallationID: result.InstallationID, HelperPath: entry.HelperPath, RuntimeHome: entry.RuntimeHome, CheckpointRoot: entry.CheckpointRoot, RuntimePath: entry.RuntimePath, FirmwarePath: entry.FirmwarePath, RuntimeSHA256: release.RuntimeSHA256, FirmwareSHA256: release.FirmwareSHA256, Image: release.MicrosandboxRef, MemoryMiB: resources.MemoryMiB, CPUs: uint8(resources.CPUs), RootDiskMiB: resources.RootDiskMiB, EnvironmentDiskMiB: resources.EnvironmentDiskMiB, Network: network, } @@ -65,7 +67,17 @@ func configureMicrosandbox(entry Native, spec sandbox.DeploymentSpec, caller *Pr } provider.workspace = options.Workspace result.Provider = provider - result.Probe = microsandboxProbe(config, resources) + baseProbe := microsandboxProbe(config, resources) + result.Probe = func(ctx context.Context) (*sandbox.CheckpointCompatibility, error) { + if err := baseProbe(ctx); err != nil { + return nil, err + } + c, err := CheckpointClass(config) + if err != nil { + return nil, err + } + return &c, nil + } result.Quiescent = caller.Quiescent result.BackendFingerprint = sandbox.BackendFingerprint("microsandbox", entry.RuntimeHome) return config, nil @@ -75,7 +87,7 @@ func configureMicrosandbox(entry Native, spec sandbox.DeploymentSpec, caller *Pr func BuildNode(c sandbox.NodeConfig, options sandbox.LocalOptions, result *sandbox.Built) (func(), error) { closeProvider := func() {} var entry Native - if sandbox.DecodeConfigurationObject(c.Native, &entry, "helper_path", "runtime_home", "runtime_path", "firmware_path", "network") != nil { + if sandbox.DecodeConfigurationObject(c.Native, &entry, "helper_path", "runtime_home", "runtime_path", "firmware_path", "network", "checkpoint_root") != nil { return closeProvider, errors.New("invalid managed microsandbox node configuration") } caller := &ProcessCaller{} @@ -96,7 +108,15 @@ func BuildNode(c sandbox.NodeConfig, options sandbox.LocalOptions, result *sandb return closeProvider, err } if options.GenerationStateDirectory != "" { - result.Probe = microsandboxGenerationProbe(config, result.Probe) + baseProbe := result.Probe + result.Probe = func(ctx context.Context) (*sandbox.CheckpointCompatibility, error) { + var compatibility *sandbox.CheckpointCompatibility + imageProbe := microsandboxGenerationProbe(config, func(ctx context.Context) error { var err error; compatibility, err = baseProbe(ctx); return err }) + if err := imageProbe(ctx); err != nil { + return nil, err + } + return compatibility, nil + } } return closeProvider, nil } diff --git a/services/core/internal/sandbox/microsandbox/node_test.go b/services/core/internal/sandbox/microsandbox/node_test.go index 082bfd5a7..8e542fda8 100644 --- a/services/core/internal/sandbox/microsandbox/node_test.go +++ b/services/core/internal/sandbox/microsandbox/node_test.go @@ -21,7 +21,13 @@ func TestMicrosandboxConstructionSelectsGenerationReadiness(t *testing.T) { } useKVM(t, true) dir := t.TempDir() - native := Native{HelperPath: filepath.Join(dir, "helper"), RuntimeHome: dir, RuntimePath: filepath.Join(dir, "msb"), FirmwarePath: filepath.Join(dir, "firmware"), Network: Network{DefaultEgress: "allow", DefaultIngress: "deny"}} + native := Native{HelperPath: filepath.Join(dir, "helper"), RuntimeHome: dir, CheckpointRoot: dir, RuntimePath: filepath.Join(dir, "msb"), FirmwarePath: filepath.Join(dir, "firmware"), Network: Network{DefaultEgress: "allow", DefaultIngress: "deny"}} + if err := os.Chmod(dir, 0700); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(dir, ".oac-checkpoint-store"), []byte(uuid.NewString()+"\n"), 0600); err != nil { + t.Fatal(err) + } raw, err := json.Marshal(native) if err != nil { t.Fatal(err) @@ -44,7 +50,7 @@ func TestMicrosandboxConstructionSelectsGenerationReadiness(t *testing.T) { if err != nil { t.Fatal(err) } - err = built.Probe(t.Context()) + _, err = built.Probe(t.Context()) closeProvider() if options.Standalone && err != nil || !options.Standalone && !errors.Is(err, sandbox.ErrRuntimeImageUnavailable) { t.Fatalf("readiness for %+v: %v", options, err) diff --git a/services/core/internal/sandbox/microsandbox/provider.go b/services/core/internal/sandbox/microsandbox/provider.go index 0b9f62800..abf5acec7 100644 --- a/services/core/internal/sandbox/microsandbox/provider.go +++ b/services/core/internal/sandbox/microsandbox/provider.go @@ -88,6 +88,13 @@ func (p *Provider) state(ctx context.Context, q Request) (State, error) { func (p *Provider) responseState(ctx context.Context, q Request, out Response) (State, error) { var e error + if out.State != nil && out.State.RestoreAttemptClosed != "" { + if q.Operation != "resume" || q.Resume == nil || !out.State.ClosesRestoreAttempt(*q.Resume) { + return State{}, ErrUnconfirmed + } + return *out.State, nil + } + if out.State == nil || ValidateCompute(p.config, q.Reference, out.State.Compute) != nil || out.State.Compute.ID == "" { return State{}, ErrUnconfirmed } diff --git a/services/core/internal/sandbox/microsandbox/provider_test.go b/services/core/internal/sandbox/microsandbox/provider_test.go index 94bfa9881..db67511af 100644 --- a/services/core/internal/sandbox/microsandbox/provider_test.go +++ b/services/core/internal/sandbox/microsandbox/provider_test.go @@ -14,7 +14,7 @@ type callerFunc func(context.Context, Request) (Response, error) func (f callerFunc) Call(c context.Context, q Request) (Response, error) { return f(c, q) } func testConfig() Config { - return Config{ + return Config{CheckpointRoot: "/private-checkpoints", InstallationID: "11111111-1111-4111-8111-111111111111", HelperPath: "/helper", RuntimeHome: "/private/msb", RuntimePath: "/private/bin/msb", FirmwarePath: "/private/lib/libkrunfw.so", RuntimeSHA256: strings.Repeat("a", 64), FirmwareSHA256: strings.Repeat("b", 64), Image: "registry/runtime@sha256:" + strings.Repeat("c", 64), MemoryMiB: 2048, CPUs: 2, RootDiskMiB: 4096, EnvironmentDiskMiB: 2048, Network: NetworkPolicy{DefaultEgress: "deny", DefaultIngress: "deny", Rules: []NetworkRule{{Action: "allow", Direction: "egress", Destination: "host"}}}, @@ -26,7 +26,7 @@ func testRef() sandbox.Reference { func testSnapshot() SnapshotIdentity { c, r := testConfig(), testRef() op := "55555555-5555-4555-8555-555555555555" - return SnapshotIdentity{Reference: SnapshotReference(c, r, op), ID: "snap_exact", Digest: "digest", CheckpointID: "checkpoint", CheckpointRoot: "root", OperationID: op, SourceName: Name(c, r, 0), SourceID: "local:4"} + return SnapshotIdentity{Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "domain", ExecutionClass: "class"}, Reference: SnapshotReference(c, r, op), ID: "snap_exact", Digest: "digest", CheckpointID: "checkpoint", CheckpointRoot: "root", OperationID: op, SourceName: Name(c, r, 0), SourceID: "local:4"} } func deadline(t *testing.T) context.Context { t.Helper() diff --git a/services/core/internal/sandbox/microsandbox/restore_settlement_test.go b/services/core/internal/sandbox/microsandbox/restore_settlement_test.go new file mode 100644 index 000000000..bbf00779b --- /dev/null +++ b/services/core/internal/sandbox/microsandbox/restore_settlement_test.go @@ -0,0 +1,41 @@ +package microsandbox + +import ( + "testing" +) + +func TestRestoreClosedReceiptRequiresExactObserveOnlyOperation(t *testing.T) { + p := &Provider{config: testConfig()} + snapshot := testSnapshot() + target := Compute{Name: Name(testConfig(), testRef(), 1), Generation: 1, RestoredFrom: &snapshot} + q := Request{Operation: "resume", Reference: testRef(), Resume: &ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: target, Snapshot: snapshot, ObserveOnly: true}} + valid := State{Compute: target, Status: "absent", RestoreAttemptClosed: q.Resume.OperationID} + for _, kind := range []string{"exact", "wrong operation", "native ID", "foreign target", "bootstrap", "snapshot", "source stopped", "not observation"} { + t.Run(kind, func(t *testing.T) { + request := q + resume := *q.Resume + request.Resume = &resume + state := valid + switch kind { + case "wrong operation": + state.RestoreAttemptClosed = "wrong" + case "native ID": + state.Compute.ID = "local:1" + case "foreign target": + state.Compute.Name = "foreign" + case "bootstrap": + state.BootstrapComplete = true + case "snapshot": + state.Snapshot = &snapshot + case "source stopped": + state.SourceStopped = true + case "not observation": + request.Resume.ObserveOnly = false + } + _, err := p.responseState(t.Context(), request, Response{State: &state}) + if (kind == "exact") != (err == nil) { + t.Fatal(kind, err) + } + }) + } +} diff --git a/services/core/internal/sandbox/microsandbox/types.go b/services/core/internal/sandbox/microsandbox/types.go index 68eb8b3d3..43442c229 100644 --- a/services/core/internal/sandbox/microsandbox/types.go +++ b/services/core/internal/sandbox/microsandbox/types.go @@ -9,7 +9,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" ) -const ProtocolVersion = 4 +const ProtocolVersion = 5 const SDKVersion = "v0.7.8" const MaxOutputBytes = 1024 * 1024 const MaxRequestBytes = 72 * 1024 * 1024 @@ -23,6 +23,7 @@ type Config struct { InstallationID string HelperPath string RuntimeHome string + CheckpointRoot string RuntimePath string FirmwarePath string RuntimeSHA256 string diff --git a/services/core/internal/sandbox/node/checkpoint_generation_test.go b/services/core/internal/sandbox/node/checkpoint_generation_test.go new file mode 100644 index 000000000..5f5bd6251 --- /dev/null +++ b/services/core/internal/sandbox/node/checkpoint_generation_test.go @@ -0,0 +1,98 @@ +package node + +import ( + "context" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/providercontract" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" + "github.com/google/uuid" + "net/http/httptest" + "strings" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +func TestStaticNodeReportsExactCheckpointGeneration(t *testing.T) { + id := identity() + compatibility := &sandbox.CheckpointCompatibility{ArtifactDomain: "private-store", ExecutionClass: "native-class"} + status := sandbox.GenerationStatus{Generation: id.DeploymentGeneration, SpecificationDigest: id.SpecificationDigest, State: "ready", Checkpoint: compatibility} + seen := false + hub := NewHub(HubOptions{Heartbeat: func(_ context.Context, _ Identity, _ string, _ uint64, health Health) error { + seen = true + if len(health.Generations) != 1 || *health.Generations[0].Checkpoint != *compatibility { + t.Fatal("qualification lost") + } + return nil + }}) + p := &peer{identity: id} + health := Health{ProviderReady: true, Generations: []sandbox.GenerationStatus{status}} + if err := hub.recordHealth(t.Context(), p, health); err != nil || !seen { + t.Fatal(err, seen) + } + for _, mutation := range []string{"generation", "digest", "readiness", "extra"} { + t.Run(mutation, func(t *testing.T) { + changed := health + changed.Generations = append([]sandbox.GenerationStatus(nil), health.Generations...) + switch mutation { + case "generation": + changed.Generations[0].Generation++ + case "digest": + changed.Generations[0].SpecificationDigest = strings.Repeat("f", 64) + case "readiness": + changed.ProviderReady = false + case "extra": + changed.Generations = append(changed.Generations, status) + } + seen = false + if err := hub.recordHealth(t.Context(), p, changed); err == nil || seen { + t.Fatal("invalid static qualification reached persistence") + } + }) + } +} + +type closedRestoreProvider struct{ fakeProvider } + +func (*closedRestoreProvider) ProviderOperations() providercontract.Operations { + return microsandbox.Operations() +} +func (*closedRestoreProvider) Resume(_ context.Context, q sandbox.ResumeRequest) (sandbox.ComputeState, error) { + return sandbox.ComputeState{Compute: q.Target, Status: "absent", RestoreAttemptClosed: q.OperationID}, nil +} +func TestClosedRestoreEvidenceSurvivesNodeTransport(t *testing.T) { + id := identity() + var credential string + hub := NewHub(HubOptions{Authenticate: func(_ context.Context, node, token string) (Identity, error) { + if node != id.NodeID || token != credential { + return Identity{}, ErrAuthentication + } + return id, nil + }, OwnerEpoch: func(context.Context) (uint64, error) { return 7, nil }}) + defer hub.Close() + server := httptest.NewServer(hub) + defer server.Close() + directory := stateDir(t) + stored, err := InitIdentity(directory, server.URL, id) + if err != nil { + t.Fatal(err) + } + credential = stored.Credential + ctx, cancel := context.WithCancel(t.Context()) + done := make(chan error, 1) + go func() { + done <- Run(ctx, AgentConfig{CoreURL: server.URL, StateDirectory: directory, Identity: id, Credential: credential, Provider: &closedRestoreProvider{}, Probe: probe}) + }() + defer func() { + cancel() + if err := <-done; err != nil { + t.Error(err) + } + }() + wait(t, func() bool { return hub.Online(id.NodeID) }) + snapshot := sandbox.SnapshotIdentity{ID: "snapshot", Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "store", ExecutionClass: "class"}} + q := sandbox.ResumeRequest{Reference: reference(), OperationID: uuid.NewString(), Snapshot: snapshot, Target: sandbox.Compute{Generation: 1, Name: "exact-target", RestoredFrom: &snapshot}, ObserveOnly: true} + result, err := hub.Proxy(id.NodeID, microsandbox.Operations(), id.DeploymentGeneration).Resume(t.Context(), q) + if err != nil || !result.ClosesRestoreAttempt(q) { + t.Fatalf("exact closed proof lost in proxy: %#v %v", result, err) + } +} diff --git a/services/core/internal/sandbox/node/generation_json_test.go b/services/core/internal/sandbox/node/generation_json_test.go index b6fef3a10..506b1066b 100644 --- a/services/core/internal/sandbox/node/generation_json_test.go +++ b/services/core/internal/sandbox/node/generation_json_test.go @@ -63,7 +63,7 @@ func TestGenerationHealthRejectsUnboundedOrAmbiguousNumbers(t *testing.T) { } func TestNodeProtocolRejectsHistoricalVersions(t *testing.T) { - for _, version := range []int{1, 2, 3, 4} { + for _, version := range []int{1, 2, 3, 4, 5} { raw, _ := json.Marshal(frame{Version: version, Type: "hello", Identity: new(Identity), Health: &Health{ObservedAt: time.Now().UTC()}}) if _, err := decodeFrame(raw); err == nil { t.Fatalf("accepted historical protocol %d", version) @@ -84,8 +84,9 @@ func TestNodeGenerationManagementIsAnExplicitCurrentCapability(t *testing.T) { if managed { f.GenerationManagement = false raw, _ = json.Marshal(f) - if _, err := decodeFrame(raw); err == nil { - t.Fatal("generation management accepted without declaration") + decoded, err := decodeFrame(raw) + if err != nil || decoded.GenerationManagement || len(decoded.Health.Generations) != 1 { + t.Fatal("static qualification implicitly enabled generation management", err) } } } diff --git a/services/core/internal/sandbox/node/generation_wire.go b/services/core/internal/sandbox/node/generation_wire.go index 11884877b..d8c83327b 100644 --- a/services/core/internal/sandbox/node/generation_wire.go +++ b/services/core/internal/sandbox/node/generation_wire.go @@ -46,6 +46,9 @@ func validateVersionFrame(f frame, size int) error { if !validGeneration(g.Generation) || !validSpecificationDigest(g.SpecificationDigest) || seen[g.Generation] || (g.State != "ready" && g.State != "preparing" && g.State != "failed") || g.State == "ready" && g.Diagnostic != "" || len(g.Diagnostic) > 64 || sandbox.NormalizeNodeDiagnostic(g.Diagnostic) != g.Diagnostic { return sandbox.ErrInvalid } + if g.Checkpoint != nil && (g.State != "ready" || g.Checkpoint.Validate() != nil) { + return sandbox.ErrInvalid + } seen[g.Generation] = true } } @@ -72,9 +75,6 @@ func validateVersionFrame(f frame, size int) error { } switch f.Type { case "hello": - if !f.GenerationManagement && f.Health != nil && f.Health.Generations != nil { - return sandbox.ErrInvalid - } if f.Identity == nil || f.Health == nil || f.Request != nil || f.Response != nil || f.Control != nil || f.Deployment != nil { return sandbox.ErrInvalid } diff --git a/services/core/internal/sandbox/node/generations.go b/services/core/internal/sandbox/node/generations.go index 83354a61a..9e44ef959 100644 --- a/services/core/internal/sandbox/node/generations.go +++ b/services/core/internal/sandbox/node/generations.go @@ -15,7 +15,7 @@ type GenerationProvider struct { Generation uint64 SpecificationDigest string Provider sandbox.SandboxProvider - Probe func(context.Context) error + Probe func(context.Context) (*sandbox.CheckpointCompatibility, error) // Quiescent reports that no helper outlived its canceled caller; nil // means always quiescent. Quiescent func() bool @@ -31,6 +31,7 @@ type GenerationManagerOptions struct { } type localGeneration struct { + checkpoint *sandbox.CheckpointCompatibility value GenerationProvider state, diagnostic string refs int @@ -164,7 +165,7 @@ func (m *GenerationManager) Statuses() []sandbox.GenerationStatus { continue } seen[key] = true - result = append(result, sandbox.GenerationStatus{Generation: key, SpecificationDigest: g.value.SpecificationDigest, State: g.state, Diagnostic: g.diagnostic}) + result = append(result, sandbox.GenerationStatus{Generation: key, SpecificationDigest: g.value.SpecificationDigest, State: g.state, Diagnostic: g.diagnostic, Checkpoint: g.checkpoint}) if key != m.target.Generation && (m.target.ServingGeneration == nil || key != *m.target.ServingGeneration) { m.statusCursor = key } @@ -360,7 +361,10 @@ func (m *GenerationManager) probeLoop() { continue } ctx, cancel := context.WithTimeout(m.ctx, 5*time.Second) - err := g.value.Probe(ctx) + checkpoint, err := g.value.Probe(ctx) + if err == nil && ((sandbox.SupportsCheckpoint(g.value.Provider) && (checkpoint == nil || checkpoint.Validate() != nil)) || (!sandbox.SupportsCheckpoint(g.value.Provider) && checkpoint != nil)) { + err = sandbox.ErrInvalid + } cancel() m.mu.Lock() g.refs-- @@ -368,9 +372,11 @@ func (m *GenerationManager) probeLoop() { if !errors.Is(err, context.Canceled) { state, diagnostic := g.state, g.diagnostic g.state = "ready" + g.checkpoint = checkpoint g.diagnostic = "" if err != nil { g.state = "failed" + g.checkpoint = nil g.diagnostic = sandbox.NodeDiagnostic(err) if errors.Is(err, sandbox.ErrRuntimeImageUnavailable) || errors.Is(err, sandbox.ErrArtifactsUnavailable) { g.repairing = true diff --git a/services/core/internal/sandbox/node/generations_test.go b/services/core/internal/sandbox/node/generations_test.go index bce4d88cd..f81a5e38f 100644 --- a/services/core/internal/sandbox/node/generations_test.go +++ b/services/core/internal/sandbox/node/generations_test.go @@ -207,7 +207,7 @@ func TestRestartRecoversExactOlderGenerationBeforeAdvertisingReadiness(t *testin probe := make(chan struct{}, 1) qualify := make(chan struct{}) m, err := NewGenerationManager(t.Context(), GenerationManagerOptions{ - Initial: []GenerationProvider{{Generation: 2, SpecificationDigest: digest, Provider: &fakeProvider{}, Probe: func(context.Context) error { return nil }}}, + Initial: []GenerationProvider{{Generation: 2, SpecificationDigest: digest, Provider: &fakeProvider{}, Probe: func(context.Context) (*sandbox.CheckpointCompatibility, error) { return nil, nil }}}, Recover: []sandbox.GenerationReference{{Generation: 1, SpecificationDigest: digest}}, Prepare: func(ctx context.Context, generation uint64, got string) (GenerationProvider, error) { if got != digest { @@ -219,16 +219,16 @@ func TestRestartRecoversExactOlderGenerationBeforeAdvertisingReadiness(t *testin case <-ctx.Done(): return GenerationProvider{}, ctx.Err() } - return GenerationProvider{Generation: generation, SpecificationDigest: got, Provider: &fakeProvider{}, Probe: func(ctx context.Context) error { + return GenerationProvider{Generation: generation, SpecificationDigest: got, Provider: &fakeProvider{}, Probe: func(ctx context.Context) (*sandbox.CheckpointCompatibility, error) { select { case probe <- struct{}{}: default: } select { case <-qualify: - return nil + return nil, nil case <-ctx.Done(): - return ctx.Err() + return nil, ctx.Err() } }}, nil }, @@ -296,12 +296,12 @@ func TestInterruptedCollectionNeverPreparesOrServesAfterRestart(t *testing.T) { probed := make(chan struct{}, 1) removes := 0 manager, err := NewGenerationManager(t.Context(), GenerationManagerOptions{ - Initial: []GenerationProvider{{Generation: 2, SpecificationDigest: digest, Provider: &fakeProvider{}, Probe: func(context.Context) error { + Initial: []GenerationProvider{{Generation: 2, SpecificationDigest: digest, Provider: &fakeProvider{}, Probe: func(context.Context) (*sandbox.CheckpointCompatibility, error) { select { case probed <- struct{}{}: default: } - return nil + return nil, nil }}}, Collect: []sandbox.GenerationReference{ref}, Prepare: func(context.Context, uint64, string) (GenerationProvider, error) { diff --git a/services/core/internal/sandbox/node/hub.go b/services/core/internal/sandbox/node/hub.go index bda1da1c4..4feb0e2f6 100644 --- a/services/core/internal/sandbox/node/hub.go +++ b/services/core/internal/sandbox/node/hub.go @@ -391,8 +391,10 @@ func (h *Hub) call(ctx context.Context, id string, q request) (response, error) } func (h *Hub) recordHealth(ctx context.Context, p *peer, health Health) error { - if !p.generationManagement && health.Generations != nil { - return sandbox.ErrInvalid + if !p.generationManagement && len(health.Generations) > 0 { + if len(health.Generations) != 1 || health.Generations[0].Generation != p.identity.DeploymentGeneration || health.Generations[0].SpecificationDigest != p.identity.SpecificationDigest || (health.Generations[0].State == "ready") != health.ProviderReady { + return sandbox.ErrInvalid + } } if p.generationManagement { if h.options.Generations == nil { diff --git a/services/core/internal/sandbox/node/proxy.go b/services/core/internal/sandbox/node/proxy.go index e064f5b38..4949c3795 100644 --- a/services/core/internal/sandbox/node/proxy.go +++ b/services/core/internal/sandbox/node/proxy.go @@ -123,6 +123,12 @@ func (p *provider) state(ctx context.Context, q request) (sandbox.ComputeState, } return sandbox.ComputeState{}, e } + if r.State != nil && r.State.RestoreAttemptClosed != "" { + if q.Operation == "resume" && q.Resume != nil && r.State.ClosesRestoreAttempt(*q.Resume) { + return *r.State, nil + } + return sandbox.ComputeState{}, sandbox.ErrComputeUnconfirmed + } if r.State == nil || r.State.Compute.ID == "" { return sandbox.ComputeState{}, sandbox.ErrComputeUnconfirmed } diff --git a/services/core/internal/sandbox/node/wire.go b/services/core/internal/sandbox/node/wire.go index d4ab402ab..e2fe64b2c 100644 --- a/services/core/internal/sandbox/node/wire.go +++ b/services/core/internal/sandbox/node/wire.go @@ -18,7 +18,7 @@ import ( "github.com/gorilla/websocket" ) -const ProtocolVersion = 5 +const ProtocolVersion = 6 const MaxControlFrameBytes = 32 * 1024 const MaxFrameBytes = 72 * 1024 * 1024 const maxPending = 32 diff --git a/services/core/internal/sandbox/node/workspace_test.go b/services/core/internal/sandbox/node/workspace_test.go index 35ab50569..cfcfe34e4 100644 --- a/services/core/internal/sandbox/node/workspace_test.go +++ b/services/core/internal/sandbox/node/workspace_test.go @@ -68,7 +68,7 @@ func (c *workspaceRefusalCaller) Call(context.Context, microsandbox.Request) (mi func TestActualWorkspaceRefusalKeepsSettledAbsenceOnWire(t *testing.T) { ref := reference() caller := new(workspaceRefusalCaller) - config := microsandbox.Config{ExternalWorkspace: true, InstallationID: "11111111-1111-4111-8111-111111111111", HelperPath: "/helper", RuntimeHome: "/runtime", RuntimePath: "/runtime/msb", FirmwarePath: "/runtime/firmware", RuntimeSHA256: strings.Repeat("a", 64), FirmwareSHA256: strings.Repeat("b", 64), Image: "image@sha256:" + strings.Repeat("c", 64), CPUs: 2, MemoryMiB: 2048, RootDiskMiB: 4096, Network: microsandbox.NetworkPolicy{DefaultEgress: "deny", DefaultIngress: "deny"}} + config := microsandbox.Config{ExternalWorkspace: true, InstallationID: "11111111-1111-4111-8111-111111111111", HelperPath: "/helper", RuntimeHome: "/runtime", CheckpointRoot: "/checkpoints", RuntimePath: "/runtime/msb", FirmwarePath: "/runtime/firmware", RuntimeSHA256: strings.Repeat("a", 64), FirmwareSHA256: strings.Repeat("b", 64), Image: "image@sha256:" + strings.Repeat("c", 64), CPUs: 2, MemoryMiB: 2048, RootDiskMiB: 4096, Network: microsandbox.NetworkPolicy{DefaultEgress: "deny", DefaultIngress: "deny"}} provider, err := microsandbox.NewWithCaller(config, caller) if err != nil { t.Fatal(err) diff --git a/services/core/internal/sandbox/sandbox_provider.go b/services/core/internal/sandbox/sandbox_provider.go index 8d311697f..e77a4c72d 100644 --- a/services/core/internal/sandbox/sandbox_provider.go +++ b/services/core/internal/sandbox/sandbox_provider.go @@ -139,10 +139,25 @@ type Compute struct { RestoredFrom *SnapshotIdentity } +// CheckpointCompatibility identifies an adapter-verified private archive domain +// and execution compatibility class. Core compares these opaque tokens only. +type CheckpointCompatibility struct { + ArtifactDomain string `json:"artifact_domain"` + ExecutionClass string `json:"execution_class"` +} + +func (c CheckpointCompatibility) Validate() error { + if c.ArtifactDomain == "" || c.ExecutionClass == "" || len(c.ArtifactDomain) > 256 || len(c.ExecutionClass) > 256 { + return ErrInvalid + } + return nil +} + // SnapshotIdentity is provider evidence from a verified full snapshot. Core // persists it unchanged and records consumption separately; it never invents // paths, checksums, native checkpoint fields, or source identity. type SnapshotIdentity struct { + Compatibility CheckpointCompatibility Reference string ID string Digest string @@ -159,8 +174,23 @@ type ComputeState struct { Status string BootstrapComplete bool Snapshot *SnapshotIdentity - SourceStopped bool + // SourceStopped proves the archive is durable and all source-local compute + // and capture obligations have settled, so another node can own restoration. + SourceStopped bool + // RestoreAttemptClosed echoes the exact Resume OperationID only after a + // never-admitted attempt is durably closed against late execution. + RestoreAttemptClosed string +} + +// ClosesRestoreAttempt validates a never-executed closure against the exact +// persisted request. Absence without this operation-bound evidence is unknown. +func (s ComputeState) ClosesRestoreAttempt(q ResumeRequest) bool { + return q.ObserveOnly && q.OperationID != "" && s.RestoreAttemptClosed == q.OperationID && s.Status == "absent" && + s.Compute.ID == "" && q.Target.ID == "" && s.Compute.Name == q.Target.Name && s.Compute.Generation == q.Target.Generation && + s.Compute.RestoredFrom != nil && q.Target.RestoredFrom != nil && *s.Compute.RestoredFrom == q.Snapshot && *q.Target.RestoredFrom == q.Snapshot && + !s.BootstrapComplete && !s.SourceStopped && s.Snapshot == nil } + type SuspendRequest struct { Reference Reference OperationID string @@ -335,7 +365,7 @@ type Built struct { SpecificationDigest string Provider SandboxProvider InstallationID, BackendFingerprint string - Probe func(context.Context) error + Probe func(context.Context) (*CheckpointCompatibility, error) // Quiescent is nil when no helper can outlive its caller. Quiescent func() bool } diff --git a/services/core/migrations/000099_checkpoint_compatibility.sql b/services/core/migrations/000099_checkpoint_compatibility.sql new file mode 100644 index 000000000..0e9b0b490 --- /dev/null +++ b/services/core/migrations/000099_checkpoint_compatibility.sql @@ -0,0 +1,6 @@ +-- +goose Up +ALTER TABLE runtime_node_generation_status ADD COLUMN checkpoint jsonb + CHECK (checkpoint IS NULL OR jsonb_typeof(checkpoint) = 'object'); + +-- +goose Down +ALTER TABLE runtime_node_generation_status DROP COLUMN checkpoint; diff --git a/services/core/tests/integration/deployment_fixture_test.go b/services/core/tests/integration/deployment_fixture_test.go index 25018ae00..2073a5b60 100644 --- a/services/core/tests/integration/deployment_fixture_test.go +++ b/services/core/tests/integration/deployment_fixture_test.go @@ -1,12 +1,14 @@ package integration import ( + "context" "testing" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/deploymentpg" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/pgunit" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/providers" ) @@ -50,3 +52,16 @@ func deploymentExecution(t testing.TB, w *Store) *deployment.ExecutionOperations } return operations } + +func fixtureNodeHeartbeat(ctx context.Context, store *Store, service *deployment.Service, nodeID, connection string, epoch uint64, health deployment.NodeHealth) error { + var generation uint64 + var digest, provider string + if err := store.pool.QueryRow(ctx, "SELECT n.deployment_generation,n.specification_digest,d.provider_kind FROM runtime_nodes n CROSS JOIN runtime_deployment d WHERE n.id=$1", nodeID).Scan(&generation, &digest, &provider); err != nil { + return err + } + var statuses []sandbox.GenerationStatus + if health.ProviderReady && provider == "microsandbox" { + statuses = []sandbox.GenerationStatus{{Generation: generation, SpecificationDigest: digest, State: "ready", Checkpoint: &sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}}} + } + return service.Heartbeat(ctx, nodeID, connection, epoch, health, statuses) +} diff --git a/services/core/tests/integration/runtime_compute_lifecycle_test.go b/services/core/tests/integration/runtime_compute_lifecycle_test.go index 84b7e6af0..4c27a8e40 100644 --- a/services/core/tests/integration/runtime_compute_lifecycle_test.go +++ b/services/core/tests/integration/runtime_compute_lifecycle_test.go @@ -76,6 +76,10 @@ func (p *fakeCheckpointProvider) Suspend(_ context.Context, q sandbox.SuspendReq defer p.mu.Unlock() state, ok := p.computes[q.Source.Name] if !ok { + if snapshot, found := p.snapshots[q.OperationID]; found && q.ObserveOnly { + p.captureObservations++ + return sandbox.ComputeState{Compute: q.Source, Snapshot: &snapshot, SourceStopped: true, Status: "suspended"}, nil + } return sandbox.ComputeState{}, sandbox.ErrNotFound } if state.Compute.ID != q.Source.ID { @@ -88,7 +92,7 @@ func (p *fakeCheckpointProvider) Suspend(_ context.Context, q sandbox.SuspendReq if _, exists := p.snapshots[q.OperationID]; exists { return sandbox.ComputeState{}, errors.New("capture replayed") } - p.snapshots[q.OperationID] = sandbox.SnapshotIdentity{Reference: "snapshot-" + q.OperationID, ID: uuid.NewString(), Digest: "verified", CheckpointID: "checkpoint", CheckpointRoot: "private", OperationID: q.OperationID, SourceGeneration: q.Source.Generation, SourceName: q.Source.Name, SourceID: q.Source.ID} + p.snapshots[q.OperationID] = sandbox.SnapshotIdentity{Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, Reference: "snapshot-" + q.OperationID, ID: uuid.NewString(), Digest: "verified", CheckpointID: "checkpoint", CheckpointRoot: "private", OperationID: q.OperationID, SourceGeneration: q.Source.Generation, SourceName: q.Source.Name, SourceID: q.Source.ID} state.Status = "paused" p.computes[q.Source.Name] = state if p.loseCapture { @@ -98,6 +102,10 @@ func (p *fakeCheckpointProvider) Suspend(_ context.Context, q sandbox.SuspendReq } if snapshot, exists := p.snapshots[q.OperationID]; exists { state.Snapshot = &snapshot + state.SourceStopped = true + state.Status = "stopped" + p.computeKills++ + delete(p.computes, q.Source.Name) } return state, nil } diff --git a/services/core/tests/integration/runtime_node_generations_test.go b/services/core/tests/integration/runtime_node_generations_test.go index 8af333c9c..20e5779fb 100644 --- a/services/core/tests/integration/runtime_node_generations_test.go +++ b/services/core/tests/integration/runtime_node_generations_test.go @@ -173,7 +173,7 @@ func TestNodeGenerationsReconnectAndV1Fallback(t *testing.T) { if err := reserveSessionPlacement(t, s, w, queued); !errors.Is(err, placement.ErrNodeUnavailable) { t.Fatal("unconfirmed connection admitted", err) } - if err = service.Heartbeat(t.Context(), node.NodeID, connection, first.OwnerEpoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err = service.Heartbeat(t.Context(), node.NodeID, connection, first.OwnerEpoch, deployment.NodeHealth{ProviderReady: true}, nil); err != nil { t.Fatal(err) } if _, err = s.CreateSession(t.Context(), uuid.NewString(), managerSessionInput(uuid.NewString())); err != nil { diff --git a/services/core/tests/integration/runtime_node_lifecycle_fixture_test.go b/services/core/tests/integration/runtime_node_lifecycle_fixture_test.go index f6803bf48..ca2573598 100644 --- a/services/core/tests/integration/runtime_node_lifecycle_fixture_test.go +++ b/services/core/tests/integration/runtime_node_lifecycle_fixture_test.go @@ -148,7 +148,7 @@ func (f *nodeIsolationFixture) online(id string) { if err := f.nodes.ConnectNode(f.t.Context(), id, connection, f.epoch); err != nil { f.t.Fatal(err) } - if err := f.nodes.Heartbeat(f.t.Context(), id, connection, f.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := fixtureNodeHeartbeat(f.t.Context(), f.store, f.nodes, id, connection, f.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { f.t.Fatal(err) } } @@ -183,7 +183,7 @@ func (f *nodeIsolationFixture) session(node string, initialize bool) (string, se f.t.Fatal(err) } for _, value := range others { - if err := f.nodes.Heartbeat(f.t.Context(), value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: false}); err != nil { + if err := fixtureNodeHeartbeat(f.t.Context(), f.store, f.nodes, value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: false}); err != nil { f.t.Fatal(err) } } @@ -192,7 +192,7 @@ func (f *nodeIsolationFixture) session(node string, initialize bool) (string, se _, err = f.placement.EnsurePlacement(f.t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: session.Environment.ID}, f.key) } for _, value := range others { - if err := f.nodes.Heartbeat(context.WithoutCancel(f.t.Context()), value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := fixtureNodeHeartbeat(context.WithoutCancel(f.t.Context()), f.store, f.nodes, value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { f.t.Fatal(err) } } diff --git a/services/core/tests/integration/runtime_nodes_test.go b/services/core/tests/integration/runtime_nodes_test.go index 4150a0681..2d83b955e 100644 --- a/services/core/tests/integration/runtime_nodes_test.go +++ b/services/core/tests/integration/runtime_nodes_test.go @@ -7,11 +7,11 @@ import ( "strings" "sync" "testing" - "time" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/runtimedevice" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" "github.com/google/uuid" "github.com/jackc/pgx/v5" @@ -34,7 +34,7 @@ func onlineManagerNode(t *testing.T, s *Store, id string) string { if err := nodes.ConnectNode(t.Context(), id, connection, managerEpoch(t, s)); err != nil { t.Fatal(err) } - if err := nodes.Heartbeat(t.Context(), id, connection, managerEpoch(t, s), deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := fixtureNodeHeartbeat(t.Context(), s, nodes, id, connection, managerEpoch(t, s), deployment.NodeHealth{ProviderReady: true}); err != nil { t.Fatal(err) } return connection @@ -72,13 +72,13 @@ func createSessionOnNode(t *testing.T, s, w *Store, tenant string, input session } nodes := deploymentService(t, s) for _, value := range others { - if err := nodes.Heartbeat(t.Context(), value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: false}); err != nil { + if err := fixtureNodeHeartbeat(t.Context(), s, nodes, value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: false}); err != nil { t.Fatal(err) } } defer func() { for _, value := range others { - if err := nodes.Heartbeat(context.WithoutCancel(t.Context()), value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := fixtureNodeHeartbeat(context.WithoutCancel(t.Context()), s, nodes, value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { t.Fatal(err) } } @@ -268,7 +268,7 @@ func TestRuntimeNodesEnrollmentAndEpoch(t *testing.T) { if next := managerEpoch(t, s); next != epoch+1 { t.Fatal(next) } - if err := nodes.Heartbeat(t.Context(), input.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); !errors.Is(err, deployment.ErrNodeCredential) { + if err := fixtureNodeHeartbeat(t.Context(), s, nodes, input.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); !errors.Is(err, deployment.ErrNodeCredential) { t.Fatal("old epoch heartbeat revived node", err) } queued, err := s.CreateSession(t.Context(), uuid.NewString(), managerSessionInput("stale")) @@ -333,7 +333,9 @@ func TestRuntimeNodesRetention(t *testing.T) { } } func TestRuntimeNodesRestoreAndCreationShareCapacity(t *testing.T) { - s, w, d := managerFixture(t, 1, 4) + s, w, view, _ := webSpecificationFixture(t, "microsandbox") + node := enrollNode(t, s, view, deployment.Capacity{MaxActive: 1, MaxRetained: 4}) + d := managerNode{InstallationID: view.InstallationID, NodeID: node.NodeID} tenant := uuid.NewString() session, err := s.CreateSession(t.Context(), tenant, managerSessionInput("first")) if err != nil { @@ -353,7 +355,9 @@ func TestRuntimeNodesRestoreAndCreationShareCapacity(t *testing.T) { if err != nil { t.Fatal(err) } - until := time.Now().Add(time.Hour) + if err := deploymentService(t, s).TouchActivity(t.Context(), allocation.TenantID, allocation.EnvironmentID); err != nil { + t.Fatal(err) + } second, err := s.CreateSession(t.Context(), tenant, managerSessionInput("second")) if err != nil { t.Fatal(err) @@ -362,7 +366,7 @@ func TestRuntimeNodesRestoreAndCreationShareCapacity(t *testing.T) { results := make(chan error, 2) go func() { <-start - _, err := deploymentExecution(t, w).SetCompute(t.Context(), allocation, "restoring", json.RawMessage(`{"target":{"id":"restore"}}`), &until, 0) + _, err := deploymentExecution(t, w).BeginRestore(t.Context(), allocation, sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, json.RawMessage(`{"target":{"id":"restore"}}`)) results <- err }() go func() { diff --git a/services/core/tests/integration/runtime_suspension_test.go b/services/core/tests/integration/runtime_suspension_test.go index 2f76e9ffc..e32df1a8c 100644 --- a/services/core/tests/integration/runtime_suspension_test.go +++ b/services/core/tests/integration/runtime_suspension_test.go @@ -10,6 +10,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/runtimedevice" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" "github.com/google/uuid" "github.com/jackc/pgx/v5/pgxpool" @@ -57,7 +58,13 @@ func runtimeSuspensionStep(t *testing.T, w *Store, owner deployment.Allocation, if owner.ComputePhase == "running" && phase == "quiescing" { idleTimeout = time.Nanosecond } - next, err := deploymentExecution(t, w).SetCompute(t.Context(), owner, phase, json.RawMessage(`{"instance":"original","snapshot":"qualified"}`), until, idleTimeout) + var next deployment.Allocation + var err error + if phase == "restoring" { + next, err = deploymentExecution(t, w).BeginRestore(t.Context(), owner, sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, json.RawMessage(`{"instance":"original","snapshot":"qualified"}`)) + } else { + next, err = deploymentExecution(t, w).SetCompute(t.Context(), owner, phase, json.RawMessage(`{"instance":"original","snapshot":"qualified"}`), until, idleTimeout) + } if err != nil { t.Fatalf("%s -> %s: %v", owner.ComputePhase, phase, err) } @@ -245,7 +252,7 @@ func TestRuntimeSuspensionRetentionAndDeletedSession(t *testing.T) { t.Fatal("snapshot retention expiry not observed", expired, err) } // Use the earlier unexpired observation to exercise expiry at the database CAS. - if _, err := deploymentExecution(t, w).SetCompute(t.Context(), retained, "restoring", json.RawMessage(`{}`), &until, 0); !errors.Is(err, deployment.ErrAllocationConflict) { + if _, err := deploymentExecution(t, w).BeginRestore(t.Context(), retained, sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, json.RawMessage(`{}`)); !errors.Is(err, deployment.ErrAllocationConflict) { t.Fatal("expired snapshot restored from stale observation", err) } if err := sessionService(t, s).DeleteSession(t.Context(), sessions.DeleteSessionCommand{TenantID: owner.TenantID, SessionID: owner.SessionID}); err != nil { @@ -258,7 +265,7 @@ func TestRuntimeSuspensionRetentionAndDeletedSession(t *testing.T) { if err != nil || !deleted.SessionDeleted || deleted.ComputeWakeRequested { t.Fatal("deleted session was woken", deleted, err) } - if _, err := deploymentExecution(t, w).SetCompute(t.Context(), deleted, "restoring", json.RawMessage(`{}`), &until, 0); !errors.Is(err, sessions.ErrNotFound) { + if _, err := deploymentExecution(t, w).BeginRestore(t.Context(), deleted, sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, json.RawMessage(`{}`)); !errors.Is(err, sessions.ErrNotFound) { t.Fatal("deleted session restored", err) } } diff --git a/services/core/tests/integration/sandbox_deployment_worker_test.go b/services/core/tests/integration/sandbox_deployment_worker_test.go index b2143dc1e..528a351c1 100644 --- a/services/core/tests/integration/sandbox_deployment_worker_test.go +++ b/services/core/tests/integration/sandbox_deployment_worker_test.go @@ -72,7 +72,7 @@ func TestSandboxDeploymentWorkerActivatesWithoutRestart(t *testing.T) { if err := deployments.ConnectNode(t.Context(), nodeID, connection, epoch); err != nil { t.Fatal(err) } - if err := deployments.Heartbeat(t.Context(), nodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := deployments.Heartbeat(t.Context(), nodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}, nil); err != nil { t.Fatal(err) } } diff --git a/services/core/tools/microsandbox-provider/README.md b/services/core/tools/microsandbox-provider/README.md index 243c6d682..0f69b881a 100644 --- a/services/core/tools/microsandbox-provider/README.md +++ b/services/core/tools/microsandbox-provider/README.md @@ -1,12 +1,12 @@ # microsandbox Sandbox Provider helper -microsandbox runs each hosted Session in its own microVM on a Linux amd64 node with KVM, and it is the Sandbox Provider that supports idle suspension. Core forwards provider operations to the node over the [node protocol](../../../../contracts/agents-api/node-generation-protocol.md); the node's adapter ([`sandbox/microsandbox`](../../internal/sandbox/microsandbox)) runs this helper once per operation. The private helper wire version is 4; mismatches are rejected. The helper links the microsandbox Go SDK v0.7.8 with its FFI library, so Core and the node program stay CGO-free Go binaries. It implements the checkpoint operations of the provider-neutral `sandbox.SandboxProvider` with full snapshots. It has no daemon, lifecycle database, scheduler or network control plane. +microsandbox runs each hosted Session in its own microVM on a Linux amd64 node with KVM, and it is the Sandbox Provider that supports idle suspension. Core forwards provider operations to the node over the [node protocol](../../../../contracts/agents-api/node-generation-protocol.md); the node's adapter ([`sandbox/microsandbox`](../../internal/sandbox/microsandbox)) runs this helper once per operation. The private helper wire version is 5; mismatches are rejected. The helper links the microsandbox Go SDK v0.7.8 with its FFI library, so Core and the node program stay CGO-free Go binaries. It implements the checkpoint operations of the provider-neutral `sandbox.SandboxProvider` with full snapshots. It has no daemon, lifecycle database, scheduler or network control plane. [Add a Sandbox Provider](../../../../docs/sandbox-provider.md) owns the provider contract. [Sandbox deployment](../../../../contracts/agents-api/sandbox-deployment.md) owns the resources, Runtime release and suspension policy; the [nodes guide](../../../../docs/getting-started/nodes.md) owns node installation, host requirements, the node's directories and its network policy. ## Installation checks -The [maintainer guide](../../../../docs/maintainers.md#runtime-images-and-helpers) builds the helper and packages the checksum-verified `msb` runtime and `libkrunfw` firmware. The node's provider configuration, written by the node installer, supplies the absolute helper, runtime and firmware paths with their SHA-256 values, the runtime home, the Runtime image reference, the saved resources and the host network policy. +The [maintainer guide](../../../../docs/maintainers.md#runtime-images-and-helpers) builds the helper and packages the checksum-verified `msb` runtime and `libkrunfw` firmware. The node's provider configuration, written by the node installer, supplies the absolute helper, runtime and firmware paths with their SHA-256 values, the runtime home, the private checkpoint root, the Runtime image reference, the saved resources and the host network policy. Before every operation the helper checks that it was built with the exact official SDK module declared in its embedded `go.mod` without a replacement, that the runtime and firmware match their hashes, that the runtime's `.msbver` ELF section reports the declared native release, that the SDK source and embedded FFI report that release, and that the SDK resolves exactly those paths with the local backend ([`main.go`](main.go)). It never installs or upgrades these files; keep them unchanged for the lifetime of the provider's backend. Ambient SDK profiles are ignored. @@ -14,6 +14,8 @@ The runtime home must be private (mode 0700), short, on local persistent storage The Runtime image is an immutable `repository@sha256:<64 lowercase hex>` reference that matches the saved Runtime release; bare image IDs and mutable tags are rejected. The node installer imports the distribution's image under that reference. To load an image by hand, `msb image load --tag repository@sha256:` must register the digest reference explicitly, with the manifest digest from `image inspect`, not the Docker image config ID. The image carries the daemon, Python 3, the native Harnesses and the shared Runtime helpers. The provider installs no registry credentials. +The checkpoint root is a separate adapter-owned directory, mode 0700, with a private `.oac-checkpoint-store` UUID marker. It is never derived from a workspace binding and must not be reachable through any guest mount. Archives contain confidential RAM and root-disk state. The marker and installation identity define the artifact domain; native release and binary hashes, Runtime image, kernel, CPU features and guest geometry define the conservative execution class. Filesystem paths, hostnames and local SDK homes do not define compatibility. Nodes can share this private root at different mount paths; the backing store must preserve durable writes, atomic rename and cross-host file locks. The installer's local default only advertises its own domain. + ## Create and bootstrap Create names the VM from a hash of the installation and allocation reference plus the compute generation; a name never serves another incarnation. It creates the VM with the saved CPUs and memory as both initial and maximum, a managed root disk of `root_disk_mib`, an owned ext4 disk of `environment_disk_mib` or the explicitly resolved external directory mounted at `/environment`, user 1000:1000, working directory `/` and the node's network policy ([`bootstrap.go`](bootstrap.go)). The resource checks run before bootstrap. Creation and restore retain the SDK's strict hostname policy enforcement; the adapter does not disable it. @@ -28,21 +30,23 @@ A helper response carries `CreateSettled` with a configuration rejection only af Core persists operation IDs, source and target generations, exact identities and snapshot evidence before it depends on them. `Initial` and `NewCompute` only construct references and allocate nothing. -- **Suspend** pauses the exact VM, captures a full snapshot under the persisted operation's derived group and member, verifies the complete checkpoint closure, then force-stops the source. Pausing alone does not release memory. A completed matching artifact is inspected instead of captured again. A full snapshot records a resource proof only after the source's limits match. -- **KillCompute** checks the precise incarnation before it stops the VM and removes its writable disks. Core calls it after it has persisted the verified snapshot, even when Suspend already stopped the source, so no chain of old writable disks grows across suspension cycles. -- **Restore** verifies the exact artifact and creates the precommitted target name. An existing target is adopted only when its immutable ID, if known, and its persisted `snapshot_parent` agree. Upstream restore defaults to public networking, so restore passes the same explicit host policy as creation, and no undeclared host resource or mount is inherited. +- **Suspend** pauses the exact VM, captures and verifies the full checkpoint closure, then exports a standalone SDK archive with its image into the private checkpoint root. The adapter synchronizes the archive and its digest, size and ownership receipt before force-stopping the exact paused source, removing its writable disks and deleting its local SDK snapshot. Only after that cleanup is synchronized does `SourceStopped` authorize transfer. A matching artifact is completed instead of captured again; a running source is never killed to finish an old capture. +- **KillCompute** fences future admission for the exact target and checks its incarnation before stopping and removing it. Under that dispatch fence the SDK serializes against native launch, waits for the runtime lifecycle lock on Kill, and rechecks the terminal incarnation under that lock on Remove. A final typed absence permits a durable cleanup receipt, including for an admitted restore whose launcher died. This proves the writer is gone, never that it did not execute. +- **Restore** requires the source-settled archive receipt, matching artifact domain and execution class, and the complete archive hash before importing through the SDK and creating the precommitted target name. An existing target is adopted only when its immutable ID, if known, and its persisted `snapshot_parent` agree. Upstream restore defaults to public networking, so restore passes the same explicit host policy as creation, and no undeclared host resource or mount is inherited. - Native restore leaves the managed root size unset because the target inherits the verified full snapshot. The helper accepts that only with a matching snapshot resource proof and the exact source and target identities, and it checks the target's CPU, memory and Environment disk before keeping the inherited proof. A missing root size never counts as unlimited capacity, and retained state is never resized. - Fresh restore and retry share one completion: verify the original artifact and resource proof, inspect the running target's resources and ancestry, persist its missing derived resource-proof label, then strictly reread the same native ID ([`restore_completion.go`](restore_completion.go)). A conflicting proof is an error. Native restore does not copy the source's ownership labels; ancestry supplies that evidence. There is no ordinary Start, replacement, disk-only restore or cold boot. - **ResumeCompute** thaws the same resident source after an aborted suspension. The pinned SDK handle method is name-based; the allocation lock and ID checks before and after the call fence every managed replacement. Manual lifecycle changes in the managed namespace are unsupported. -- **DeleteSnapshot** accepts only the derived operation selector and the matching full artifact identity, never an arbitrary path. Core owns retention, consumed snapshot generations and cleanup order. A checkpoint never rolls back work admitted after its first restore. +- **DeleteSnapshot** accepts only the derived operation selector and matching artifact identity. It fences delayed publication and restoration with a durable deletion receipt before deleting the local SDK snapshot and portable archive. An unsettled admitted restore blocks deletion. The small ownership tombstone remains; Core owns retention and cleanup order. + +After a lost response Core uses `ObserveOnly`. It never starts a capture or restore. For an existing verified capture it may complete archive publication and exact paused-source cleanup, but never captures again or kills a running source. Source termination is recorded before local removal so a lost cleanup response can be settled from the archive receipt. A valid ownership receipt remains discoverable when archive bytes are missing or corrupt: observation returns its snapshot identity with unknown source state and `SourceStopped=false`, so ordinary explicit cleanup can proceed. Restore still requires complete archive verification. An interrupted restore may finish the derived proof on the exact running target, but never restarts a stopped target or restores again. -After a lost response Core uses `ObserveOnly`. It never starts a capture or restore, and observing a suspend operation never kills its source. Observation checks artifact integrity and source ownership independently of resource checks, so resource drift cannot hide a retained artifact from cleanup; Core can persist recovered snapshot evidence before KillCompute. If the artifact is absent but the exact source is still running or paused with a settled bootstrap, observation returns the source without a snapshot and Core can abort the suspension; thawing and further execution still require the resource checks. For an interrupted restore, `ObserveOnly` may finish the missing resource proof on the exact target but never restarts a stopped target, changes resources or restores again. Missing state never authorizes a replay. +Before native import or restore, the helper synchronizes an admission record bound to the request operation, snapshot and target. Observation of a never-admitted operation writes a permanent closed record and returns that exact `RestoreAttemptClosed`; late dispatch is rejected. After admission, helper death or native absence alone never establishes no effect. Successful resume requires an observable exact running target. A failed or uncertain admitted restore retains its artifact and owner until explicit deletion or the original retention deadline; the normal cleanup operation uses the native lifecycle fence and records completion. It never shortens retention or substitutes a cold restart. GetCompute, commands, cleanup and the next suspension verify restored provenance from the persisted VM configuration after the consumed artifact is deleted. ## Locks and commands -A helper holds a per-allocation lock, under `oac-locks/` in the runtime home, until its SDK call actually settles. Core's response deadline neither kills the helper nor cancels its FFI wait, because cancelling the wait does not prove that the native mutation stopped. On a timeout Core keeps an unknown operation and observes it; a later helper cannot pass the surviving lock holder. A stuck owner needs operator investigation, not lock deletion or another Create. +A helper holds a per-allocation lock in the explicitly configured checkpoint root, until its SDK call actually settles. Core's response deadline neither kills the helper nor cancels its FFI wait, because cancelling the wait does not prove that the native mutation stopped. On a timeout Core keeps an unknown operation and observes it; a later helper cannot pass the surviving lock holder. A stuck owner needs operator investigation, not lock deletion or another Create. The same lock file fences a Create helper that has not yet acquired its lock, including across a node restart. An empty file permits initial Create admission; `closed\n` permanently forbids it, and any other nonempty content fails closed. Only `initial_info` observing typed native absence writes this marker while holding the lock, then synchronizes the file and its directories before returning the receipt. Every Create checks the marker under the same lock before invoking the SDK. Read, write or synchronization errors never establish settlement. The marker is never removed, and neither missing compute nor this receipt permits retrying Create. diff --git a/services/core/tools/microsandbox-provider/archive.go b/services/core/tools/microsandbox-provider/archive.go new file mode 100644 index 000000000..f3d159759 --- /dev/null +++ b/services/core/tools/microsandbox-provider/archive.go @@ -0,0 +1,467 @@ +//go:build linux + +package main + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "io" + "os" + "path/filepath" + "reflect" + "strings" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + wire "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" + sdk "github.com/superradcompany/microsandbox/sdk/go" +) + +// This adapter-private namespace is never a guest mount. The allocation lock +// serializes archive publication, admission and deletion across all its nodes. +type archiveReceipt struct { + Snapshot wire.SnapshotIdentity `json:"snapshot"` + SHA256 string `json:"sha256"` + Size int64 `json:"size"` + SourceTerminated bool `json:"source_terminated"` + SourceStopped bool `json:"source_stopped"` + Deleted bool `json:"deleted"` +} + +func (b backend) storeDirectory() string { + return filepath.Join(b.q.Config.CheckpointRoot, wire.Name(b.q.Config, b.q.Reference, 0)) +} +func (b backend) archiveDirectory(operation string) string { + return filepath.Join(b.storeDirectory(), "s-"+operation) +} +func privateDirectory(path string) error { + if err := os.Mkdir(path, 0700); err != nil && !errors.Is(err, os.ErrExist) { + return err + } + st, err := os.Lstat(path) + if err != nil { + return err + } + if !st.IsDir() || st.Mode().Perm() != 0700 { + return sandbox.ErrOwnership + } + return nil +} +func syncDirectory(path string) error { + f, e := os.Open(path) + if e != nil { + return e + } + defer f.Close() + return f.Sync() +} +func durableJSON(path string, value any) error { + data, e := json.Marshal(value) + if e != nil { + return e + } + f, e := os.CreateTemp(filepath.Dir(path), ".publish-") + if e != nil { + return e + } + defer os.Remove(f.Name()) + if _, e = f.Write(data); e == nil { + e = f.Sync() + } + closeErr := f.Close() + if e != nil { + return e + } + if closeErr != nil { + return closeErr + } + if e = os.Rename(f.Name(), path); e != nil { + return e + } + return syncDirectory(filepath.Dir(path)) +} +func readPrivateJSON(path string, value any) error { + st, e := os.Lstat(path) + if e != nil { + return e + } + if !st.Mode().IsRegular() || st.Mode().Perm() != 0600 { + return sandbox.ErrOwnership + } + real, e := filepath.EvalSymlinks(path) + if e != nil { + return e + } + if real != path { + return sandbox.ErrOwnership + } + f, e := os.Open(path) + if e != nil { + return e + } + defer f.Close() + d := json.NewDecoder(io.LimitReader(f, 65537)) + d.DisallowUnknownFields() + if e = d.Decode(value); e != nil { + return e + } + var extra any + if d.Decode(&extra) != io.EOF { + return sandbox.ErrOwnership + } + return nil +} +func (b backend) receipt(operation string) (archiveReceipt, error) { + var r archiveReceipt + e := readPrivateJSON(filepath.Join(b.archiveDirectory(operation), "receipt.json"), &r) + if e == nil && (r.Snapshot.OperationID != operation || wire.ValidateSnapshot(b.q.Config, b.q.Reference, r.Snapshot) != nil) { + e = sandbox.ErrOwnership + } + if e == nil && !r.Deleted { + decoded, err := hex.DecodeString(r.SHA256) + if err != nil || len(decoded) != 32 || r.Size <= 0 || (r.SourceStopped && !r.SourceTerminated) { + e = sandbox.ErrOwnership + } + } + return r, e +} +func archiveDigest(path string) (string, int64, error) { + real, e := filepath.EvalSymlinks(path) + if e != nil { + return "", 0, e + } + if real != path { + return "", 0, sandbox.ErrOwnership + } + st, e := os.Lstat(path) + if e != nil { + return "", 0, e + } + if !st.Mode().IsRegular() || st.Mode().Perm() != 0600 { + return "", 0, sandbox.ErrOwnership + } + f, e := os.Open(path) + if e != nil { + return "", 0, e + } + defer f.Close() + h := sha256.New() + n, e := io.Copy(h, f) + return hex.EncodeToString(h.Sum(nil)), n, e +} +func (b backend) publishArchive(ctx context.Context, a *sdk.SnapshotArtifact, s wire.SnapshotIdentity) (archiveReceipt, error) { + prior, e := b.receipt(s.OperationID) + if e == nil { + if prior.Snapshot != s || prior.Deleted { + return prior, sandbox.ErrOwnership + } + return prior, nil + } + if !errors.Is(e, os.ErrNotExist) { + return prior, e + } + dir := b.archiveDirectory(s.OperationID) + if e = privateDirectory(dir); e != nil { + return prior, e + } + if e = syncDirectory(b.storeDirectory()); e != nil { + return prior, e + } + archive := filepath.Join(dir, "archive.msb") + // A publication interrupted before its receipt can be rebuilt from the exact + // verified native artifact. The shared allocation lock excludes consumers. + if e = os.Remove(archive); e != nil && !errors.Is(e, os.ErrNotExist) { + return prior, e + } + if e = a.SaveTo(ctx, archive, sdk.SnapshotSaveOptions{WithImage: true}); e != nil { + return prior, e + } + if e = os.Chmod(archive, 0600); e != nil { + return prior, e + } + f, e := os.Open(archive) + if e != nil { + return prior, e + } + e = f.Sync() + f.Close() + if e != nil { + return prior, e + } + digest, size, e := archiveDigest(archive) + if e != nil { + return prior, e + } + r := archiveReceipt{Snapshot: s, SHA256: digest, Size: size} + return r, durableJSON(filepath.Join(dir, "receipt.json"), r) +} +func (b backend) importArchive(ctx context.Context, want wire.SnapshotIdentity) (*sdk.SnapshotArtifact, error) { + r, e := b.receipt(want.OperationID) + if e != nil { + return nil, e + } + if r.Deleted || !r.SourceStopped || r.Snapshot != want { + return nil, sandbox.ErrOwnership + } + class, e := wire.CheckpointClass(b.q.Config) + if e != nil { + return nil, e + } + if class != want.Compatibility { + return nil, sandbox.ErrInvalid + } + path := filepath.Join(b.archiveDirectory(want.OperationID), "archive.msb") + hash, size, e := archiveDigest(path) + if e != nil { + return nil, e + } + if hash != r.SHA256 || size != r.Size { + return nil, sandbox.ErrOwnership + } + artifact, e := b.verifiedSnapshot(ctx, want) + if sdk.IsKind(e, sdk.ErrSnapshotNotFound) { + if _, e = sdk.Snapshot.LoadWithOptions(ctx, path, sdk.SnapshotLoadOptions{Group: wire.Name(b.q.Config, b.q.Reference, 0)}); e != nil { + return nil, e + } + return b.verifiedSnapshot(ctx, want) + } + return artifact, e +} +func (b backend) deleteArchive(ctx context.Context, want wire.SnapshotIdentity) error { + return b.deleteArchiveWithNative(want, func() error { + _, err := b.verifiedSnapshot(ctx, want) + if err == nil { + err = sdk.Snapshot.Remove(ctx, want.Reference, false) + } + if sdk.IsKind(err, sdk.ErrSnapshotNotFound) { + return nil + } + return err + }) +} + +// Native artifact removal must settle before the private archive is unlinked. +func (b backend) deleteArchiveWithNative(want wire.SnapshotIdentity, removeNative func() error) error { + if e := b.requireSettledRestores(nil, &want); e != nil { + return e + } + r, e := b.receipt(want.OperationID) + if errors.Is(e, os.ErrNotExist) { + r = archiveReceipt{Snapshot: want} + if e = privateDirectory(b.archiveDirectory(want.OperationID)); e != nil { + return e + } + if e = syncDirectory(b.storeDirectory()); e != nil { + return e + } + } else if e != nil { + return e + } + if r.Snapshot != want { + return sandbox.ErrOwnership + } + // Fence consumers before removing either local or portable artifact. Keep the + // small tombstone, including its ownership identity, against delayed helpers. + r.Deleted = true + if e = durableJSON(filepath.Join(b.archiveDirectory(want.OperationID), "receipt.json"), r); e != nil { + return e + } + if e = removeNative(); e != nil { + return e + } + if e = os.Remove(filepath.Join(b.archiveDirectory(want.OperationID), "archive.msb")); e != nil && !errors.Is(e, os.ErrNotExist) { + return e + } + return syncDirectory(b.archiveDirectory(want.OperationID)) +} + +type restoreAdmission struct { + OperationID string + Snapshot wire.SnapshotIdentity + Target wire.Compute + Closed bool + CompletedID string + Cleaned bool +} + +func (b backend) admitRestore(q wire.ResumeRequest) (bool, error) { + if _, e := os.Lstat(filepath.Join(b.storeDirectory(), "k-"+q.Target.Name+".json")); e == nil { + return false, wire.ErrUnconfirmed + } else if !errors.Is(e, os.ErrNotExist) { + return false, e + } + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + want := restoreAdmission{OperationID: q.OperationID, Snapshot: q.Snapshot, Target: q.Target} + want.Target.ID = "" + var prior restoreAdmission + e := readPrivateJSON(path, &prior) + if e == nil { + closed := prior.Closed + prior.Closed = false + prior.CompletedID = "" + cleaned := prior.Cleaned + prior.Cleaned = false + if !reflect.DeepEqual(prior, want) { + return false, sandbox.ErrOwnership + } + if closed { + return true, nil + } + if cleaned { + return false, wire.ErrUnconfirmed + } + if !q.ObserveOnly { + return false, wire.ErrUnconfirmed + } + return false, nil + } + if !errors.Is(e, os.ErrNotExist) { + return false, e + } + if q.Target.ID != "" { + return false, sandbox.ErrOwnership + } + want.Closed = q.ObserveOnly + if e = durableJSON(path, want); e != nil { + return false, e + } + return want.Closed, nil +} + +// Completion records the incarnation whose native restore has become observable +// and whose full provenance and running state have been verified. This is +// execution evidence; cleanup separately fences dispatch and uses native locks. +func (b backend) completeRestore(ctx context.Context, q wire.ResumeRequest, target wire.Compute) (wire.State, error) { + state, e := b.finishRestore(ctx, target) + if e != nil { + return state, e + } + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + var record restoreAdmission + if e = readPrivateJSON(path, &record); e != nil { + return state, e + } + if record.Closed || record.Snapshot != q.Snapshot || record.Target.Name != target.Name || record.Target.Generation != target.Generation { + return state, sandbox.ErrOwnership + } + if record.CompletedID != "" && record.CompletedID != state.Compute.ID { + return state, sandbox.ErrOwnership + } + record.CompletedID = state.Compute.ID + return state, durableJSON(path, record) +} +func (b backend) requireSettledRestores(target *wire.Compute, snapshot *wire.SnapshotIdentity) error { + files, e := os.ReadDir(b.storeDirectory()) + if e != nil { + return e + } + for _, entry := range files { + if !strings.HasPrefix(entry.Name(), "r-") || !strings.HasSuffix(entry.Name(), ".json") { + continue + } + var r restoreAdmission + if e = readPrivateJSON(filepath.Join(b.storeDirectory(), entry.Name()), &r); e != nil { + return e + } + if entry.Name() != "r-"+r.OperationID+".json" { + return sandbox.ErrOwnership + } + if r.Closed || r.Cleaned { + continue + } + if target != nil && (r.Target.Name != target.Name || r.Target.Generation != target.Generation) { + continue + } + if snapshot != nil && r.Snapshot != *snapshot { + continue + } + if r.CompletedID == "" { + return wire.ErrUnconfirmed + } + if target != nil && target.ID == "" { + target.ID = r.CompletedID + } + if target != nil && target.ID != "" && target.ID != r.CompletedID { + return sandbox.ErrOwnership + } + } + return nil +} + +func (b backend) verifyArchive(r archiveReceipt) error { + hash, size, e := archiveDigest(filepath.Join(b.archiveDirectory(r.Snapshot.OperationID), "archive.msb")) + if e != nil { + return e + } + if hash != r.SHA256 || size != r.Size { + return sandbox.ErrOwnership + } + return nil +} + +// The native transition/lifecycle locks fence an admitted launcher and its +// detached child. This receipt means cleanup completed, never "did not run". +func (b backend) recordRestoreCleanup(target wire.Compute) error { + files, e := os.ReadDir(b.storeDirectory()) + if e != nil { + return e + } + for _, entry := range files { + if !strings.HasPrefix(entry.Name(), "r-") || !strings.HasSuffix(entry.Name(), ".json") { + continue + } + path := filepath.Join(b.storeDirectory(), entry.Name()) + var r restoreAdmission + if e = readPrivateJSON(path, &r); e != nil { + return e + } + if entry.Name() != "r-"+r.OperationID+".json" { + return sandbox.ErrOwnership + } + if r.Target.Name != target.Name || r.Target.Generation != target.Generation || r.Closed || r.Cleaned { + continue + } + if target.ID != "" && r.CompletedID != "" && target.ID != r.CompletedID { + return sandbox.ErrOwnership + } + r.Cleaned = true + if e = durableJSON(path, r); e != nil { + return e + } + } + return nil +} + +func (b backend) pinRestoreCleanup(target *wire.Compute) error { + files, e := os.ReadDir(b.storeDirectory()) + if e != nil { + return e + } + for _, entry := range files { + if !strings.HasPrefix(entry.Name(), "r-") || !strings.HasSuffix(entry.Name(), ".json") { + continue + } + var r restoreAdmission + if e = readPrivateJSON(filepath.Join(b.storeDirectory(), entry.Name()), &r); e != nil { + return e + } + if entry.Name() != "r-"+r.OperationID+".json" { + return sandbox.ErrOwnership + } + if r.Target.Name != target.Name || r.Target.Generation != target.Generation || r.Closed { + continue + } + if target.RestoredFrom == nil || *target.RestoredFrom != r.Snapshot { + return sandbox.ErrOwnership + } + if r.CompletedID != "" { + if target.ID != "" && target.ID != r.CompletedID { + return sandbox.ErrOwnership + } + target.ID = r.CompletedID + } + } + return nil +} diff --git a/services/core/tools/microsandbox-provider/backend.go b/services/core/tools/microsandbox-provider/backend.go index a9e6d9d55..cc27005c6 100644 --- a/services/core/tools/microsandbox-provider/backend.go +++ b/services/core/tools/microsandbox-provider/backend.go @@ -5,6 +5,7 @@ package main import ( "context" "errors" + "path/filepath" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" wire "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" @@ -51,14 +52,7 @@ func (b backend) run(ctx context.Context) (wire.Response, error) { s, e := b.resume(ctx, *b.q.Resume) return wire.Response{State: &s}, e case "delete_snapshot": - _, e := b.verifiedSnapshot(ctx, *b.q.Snapshot) - if sdk.IsKind(e, sdk.ErrSnapshotNotFound) { - return wire.Response{}, nil - } - if e != nil { - return wire.Response{}, e - } - return wire.Response{}, sdk.Snapshot.Remove(ctx, b.q.Snapshot.Reference, false) + return wire.Response{}, b.deleteArchive(ctx, *b.q.Snapshot) } return wire.Response{}, sandbox.ErrInvalid } @@ -81,13 +75,23 @@ func (b backend) inspectOwned(ctx context.Context, c wire.Compute) (*sdk.Sandbox return h, state, e } func (b backend) kill(ctx context.Context, c wire.Compute) error { + // Close dispatch before consulting native state. An admitted helper has either + // released this allocation flock or died; the SDK's transition and inherited + // lifecycle locks fence any surviving native launcher/runtime. + if e := durableJSON(filepath.Join(b.storeDirectory(), "k-"+c.Name+".json"), c); e != nil { + return e + } + if e := b.pinRestoreCleanup(&c); e != nil { + return e + } h, _, e := b.inspectOwned(ctx, c) if sdk.IsKind(e, sdk.ErrSandboxNotFound) { - return nil + return b.recordRestoreCleanup(c) } if e != nil { return e } + c.ID = h.ID() if e = h.Kill(ctx); e != nil { return e } @@ -96,13 +100,14 @@ func (b backend) kill(ctx context.Context, c wire.Compute) error { } _, e = sdk.GetSandbox(ctx, c.Name) if sdk.IsKind(e, sdk.ErrSandboxNotFound) { - return nil + return b.recordRestoreCleanup(c) } if e == nil { return errors.New("compute removal unconfirmed") } return e } + func (b backend) network() *sdk.NetworkConfig { n := b.q.Config.Network result := &sdk.NetworkConfig{DefaultEgress: sdk.PolicyAction(n.DefaultEgress), DefaultIngress: sdk.PolicyAction(n.DefaultIngress)} diff --git a/services/core/tools/microsandbox-provider/lock_test.go b/services/core/tools/microsandbox-provider/lock_test.go index 46c93595f..02e0ab5eb 100644 --- a/services/core/tools/microsandbox-provider/lock_test.go +++ b/services/core/tools/microsandbox-provider/lock_test.go @@ -16,8 +16,8 @@ import ( ) func TestAllocationLockSurvivesCallerDeadlineUntilExplicitSettlement(t *testing.T) { - home := t.TempDir() - q := wire.Request{Config: wire.Config{RuntimeHome: home}, Deadline: time.Now().Add(time.Second)} + home := checkpointFixture(t) + q := wire.Request{Config: wire.Config{RuntimeHome: home, CheckpointRoot: home}, Deadline: time.Now().Add(time.Second)} release, e := allocationLock(q) if e != nil { t.Fatal(e) @@ -39,12 +39,12 @@ func TestAllocationLockSurvivesCallerDeadlineUntilExplicitSettlement(t *testing. release.Close() } func TestLockDirectoryCannotRedirectIntoAnotherHome(t *testing.T) { - home := t.TempDir() + home := checkpointFixture(t) foreign := t.TempDir() - if e := os.Symlink(foreign, filepath.Join(home, "oac-locks")); e != nil { + if e := os.Symlink(foreign, filepath.Join(home, wire.Name(wire.Config{}, sandbox.Reference{}, 0))); e != nil { t.Fatal(e) } - _, e := allocationLock(wire.Request{Config: wire.Config{RuntimeHome: home}, Deadline: time.Now().Add(time.Second)}) + _, e := allocationLock(wire.Request{Config: wire.Config{RuntimeHome: home, CheckpointRoot: home}, Deadline: time.Now().Add(time.Second)}) if e == nil { t.Fatal("symlink lock directory accepted") } @@ -56,6 +56,7 @@ func initialInfoRequest(t *testing.T) wire.Request { config.InstallationID = "11111111-1111-4111-8111-111111111111" config.HelperPath, config.RuntimePath, config.FirmwarePath = "/helper", "/runtime", "/firmware" config.RuntimeHome = t.TempDir() + config.CheckpointRoot = checkpointFixture(t) config.RuntimeSHA256, config.FirmwareSHA256 = strings.Repeat("a", 64), strings.Repeat("b", 64) config.Network = wire.NetworkPolicy{DefaultEgress: "deny", DefaultIngress: "deny"} ref := sandbox.Reference{TenantID: "22222222-2222-4222-8222-222222222222", EnvironmentID: "33333333-3333-4333-8333-333333333333", AllocationID: "44444444-4444-4444-8444-444444444444"} @@ -193,3 +194,15 @@ func TestInvalidOrUnwritableAdmissionStateFailsClosed(t *testing.T) { } }) } + +func checkpointFixture(t *testing.T) string { + t.Helper() + root := t.TempDir() + if e := os.Chmod(root, 0700); e != nil { + t.Fatal(e) + } + if e := os.WriteFile(filepath.Join(root, ".oac-checkpoint-store"), []byte("55555555-5555-4555-8555-555555555555\n"), 0600); e != nil { + t.Fatal(e) + } + return root +} diff --git a/services/core/tools/microsandbox-provider/main.go b/services/core/tools/microsandbox-provider/main.go index b7dcb9a5a..f6220baae 100644 --- a/services/core/tools/microsandbox-provider/main.go +++ b/services/core/tools/microsandbox-provider/main.go @@ -58,6 +58,13 @@ func serve(input io.Reader) wire.Response { out.ErrorCode = "unconfirmed" return out } + if q.Workspace != nil { + real, err := filepath.EvalSymlinks(q.Workspace.Path) + if err != nil || real != q.Workspace.Path { + out.ErrorCode = "ownership" + return out + } + } lock, err := allocationLock(q) if err != nil { out.ErrorCode = "unconfirmed" @@ -173,7 +180,10 @@ func (g *allocationGuard) settleInitialAbsence(q wire.Request, nativeErr error) } func allocationLock(q wire.Request) (*allocationGuard, error) { - dir := filepath.Join(q.Config.RuntimeHome, "oac-locks") + if _, e := wire.CheckpointStore(q.Config); e != nil { + return nil, e + } + dir := filepath.Join(q.Config.CheckpointRoot, wire.Name(q.Config, q.Reference, 0)) if e := os.MkdirAll(dir, 0700); e != nil { return nil, e } @@ -184,6 +194,9 @@ func allocationLock(q wire.Request) (*allocationGuard, error) { if !st.IsDir() || st.Mode().Perm() != 0700 { return nil, sandbox.ErrOwnership } + if e := syncDirectory(q.Config.CheckpointRoot); e != nil { + return nil, e + } name := filepath.Join(dir, wire.Name(q.Config, q.Reference, 0)+".lock") fd, e := syscall.Open(name, syscall.O_CREAT|syscall.O_RDWR|syscall.O_NOFOLLOW|syscall.O_CLOEXEC, 0600) if e != nil { diff --git a/services/core/tools/microsandbox-provider/snapshot.go b/services/core/tools/microsandbox-provider/snapshot.go index 6cb6ee539..6d5ebb79a 100644 --- a/services/core/tools/microsandbox-provider/snapshot.go +++ b/services/core/tools/microsandbox-provider/snapshot.go @@ -5,7 +5,10 @@ package main import ( "context" "encoding/json" + "errors" "fmt" + "os" + "path/filepath" "strconv" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" @@ -49,7 +52,8 @@ func (b backend) inspectSnapshot(ctx context.Context, operation string, source w if report.Digest != artifact.Digest() || report.Checkpoint == nil || report.Checkpoint.Root == "" || report.Checkpoint.Kind == "" || state.Checkpoint == nil || state.Checkpoint.CheckpointID == "" { return nil, wire.SnapshotIdentity{}, sandbox.ErrOwnership } - identity := wire.SnapshotIdentity{Reference: selector, ID: artifact.ID(), Digest: artifact.Digest(), CheckpointID: state.Checkpoint.CheckpointID, CheckpointRoot: report.Checkpoint.Root, OperationID: operation, SourceName: source.Name, SourceID: source.ID, SourceGeneration: source.Generation} + compatibility := sandbox.CheckpointCompatibility{ArtifactDomain: artifact.Labels()["io.oac.artifact_domain"], ExecutionClass: artifact.Labels()["io.oac.execution_class"]} + identity := wire.SnapshotIdentity{Compatibility: compatibility, Reference: selector, ID: artifact.ID(), Digest: artifact.Digest(), CheckpointID: state.Checkpoint.CheckpointID, CheckpointRoot: report.Checkpoint.Root, OperationID: operation, SourceName: source.Name, SourceID: source.ID, SourceGeneration: source.Generation} if wire.ValidateSnapshot(b.q.Config, b.q.Reference, identity) != nil { return nil, wire.SnapshotIdentity{}, sandbox.ErrOwnership } @@ -66,6 +70,16 @@ func (b backend) verifiedSnapshot(ctx context.Context, want wire.SnapshotIdentit return a, nil } func (b backend) suspend(ctx context.Context, q wire.SuspendRequest) (wire.State, error) { + receipt, receiptErr := b.receipt(q.OperationID) + if receiptErr == nil { + if receipt.Deleted || receipt.Snapshot.SourceName != q.Source.Name || receipt.Snapshot.SourceID != q.Source.ID || receipt.Snapshot.SourceGeneration != q.Source.Generation || (q.Snapshot != nil && *q.Snapshot != receipt.Snapshot) { + return wire.State{}, sandbox.ErrOwnership + } + return b.settleSource(ctx, q, receipt) + } else if !errors.Is(receiptErr, os.ErrNotExist) { + return wire.State{}, receiptErr + } + // The host allocation flock spans pause, capture, verification and source kill. // A surviving completed artifact is observed, never overwritten or recaptured. artifact, snap, e := b.inspectSnapshot(ctx, q.OperationID, q.Source) @@ -97,6 +111,12 @@ func (b backend) suspend(ctx context.Context, q wire.SuspendRequest) (wire.State return wire.State{}, err } labels := b.snapshotLabels(q.OperationID, q.Source) + compatibility, err := wire.CheckpointClass(b.q.Config) + if err != nil { + return wire.State{}, err + } + labels["io.oac.artifact_domain"] = compatibility.ArtifactDomain + labels["io.oac.execution_class"] = compatibility.ExecutionClass var captured struct { Labels map[string]string `json:"labels"` } @@ -120,23 +140,37 @@ func (b backend) suspend(ctx context.Context, q wire.SuspendRequest) (wire.State if q.Snapshot != nil && snap != *q.Snapshot { return wire.State{}, sandbox.ErrOwnership } - if q.ObserveOnly { - // A completed operation receipt remains observable for cleanup even if - // its source or resource proof is no longer qualified for execution. - return observeCapturedSnapshot(q.Source, snap, func(source wire.Compute) (wire.State, error) { - _, state, err := b.inspectOwned(ctx, source) - return state, err - }) - } if e = qualifySnapshotResources(b.q.Config, artifact.Labels()); e != nil { return wire.State{}, e } - h, _, e := b.inspect(ctx, q.Source) + receipt, e = b.publishArchive(ctx, artifact, snap) + if e != nil { + return wire.State{}, e + } + return b.settleSource(ctx, q, receipt) +} + +func (b backend) settleSource(ctx context.Context, q wire.SuspendRequest, r archiveReceipt) (wire.State, error) { + if e := b.verifyArchive(r); e != nil { + if q.ObserveOnly { + // Ownership comes from the durable receipt, independently of archive + // usability. Expose cleanup identity without authorizing source transfer, + // claiming its current state, or weakening the restore integrity check. + return wire.State{Compute: q.Source, Status: "unknown", Snapshot: &r.Snapshot}, nil + } + return wire.State{}, e + } + if r.SourceStopped { + return wire.State{Compute: q.Source, Status: "suspended", BootstrapComplete: true, Snapshot: &r.Snapshot, SourceStopped: true}, nil + } + h, state, e := b.inspectOwned(ctx, q.Source) if e != nil && !sdk.IsKind(e, sdk.ErrSandboxNotFound) { return wire.State{}, e } - if e == nil { - // Graceful stop could run the captured source after the checkpoint. + if e == nil && !terminal(sdk.SandboxStatus(state.Status)) { + if state.Status != "paused" { + return wire.State{}, wire.ErrUnconfirmed + } if e = h.Kill(ctx); e != nil { return wire.State{}, e } @@ -147,12 +181,62 @@ func (b backend) suspend(ctx context.Context, q wire.SuspendRequest) (wire.State if !terminal(observed.Status()) { return wire.State{}, fmt.Errorf("source termination unconfirmed") } + } else if sdk.IsKind(e, sdk.ErrSandboxNotFound) && !r.SourceTerminated { + return wire.State{}, wire.ErrUnconfirmed + } + if !r.SourceTerminated { + r.SourceTerminated = true + if e = durableJSON(filepath.Join(b.archiveDirectory(q.OperationID), "receipt.json"), r); e != nil { + return wire.State{}, e + } + } + if h != nil { + if e = h.Remove(ctx); e != nil { + return wire.State{}, e + } + } + if _, e = sdk.GetSandbox(ctx, q.Source.Name); !sdk.IsKind(e, sdk.ErrSandboxNotFound) { + return wire.State{}, wire.ErrUnconfirmed } - return wire.State{Compute: q.Source, Status: "suspended", BootstrapComplete: true, Snapshot: &snap, SourceStopped: true}, nil + if e = sdk.Snapshot.Remove(ctx, r.Snapshot.Reference, false); e != nil && !sdk.IsKind(e, sdk.ErrSnapshotNotFound) { + return wire.State{}, e + } + r.SourceStopped = true + if e = durableJSON(filepath.Join(b.archiveDirectory(q.OperationID), "receipt.json"), r); e != nil { + return wire.State{}, e + } + return wire.State{Compute: q.Source, Status: "suspended", BootstrapComplete: true, Snapshot: &r.Snapshot, SourceStopped: true}, nil } + func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, error) { - // Verify the exact full artifact before either adopting or creating a target. - artifact, e := b.verifiedSnapshot(ctx, q.Snapshot) + // No managed target may predate its exact admission journal. + if _, err := os.Lstat(filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json")); errors.Is(err, os.ErrNotExist) { + _, err = sdk.GetSandbox(ctx, q.Target.Name) + if err == nil { + return wire.State{}, sandbox.ErrOwnership + } + if !sdk.IsKind(err, sdk.ErrSandboxNotFound) { + return wire.State{}, err + } + } else if err != nil { + return wire.State{}, err + } + + closed, e := b.admitRestore(q) + if e != nil { + return wire.State{}, e + } + if closed { + if !q.ObserveOnly { + return wire.State{}, wire.ErrUnconfirmed + } + target := q.Target + target.ID = "" + return wire.State{Compute: target, Status: "absent", RestoreAttemptClosed: q.OperationID}, nil + } + // Admission is durable before any native import/restore operation. A prior + // admitted operation with no visible target remains unknown, never replayed. + artifact, e := b.importArchive(ctx, q.Snapshot) if e != nil { return wire.State{}, e } @@ -164,7 +248,7 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, } _, _, e = b.inspectOwned(ctx, q.Target) if e == nil { - return b.finishRestore(ctx, q.Target) + return b.completeRestore(ctx, q, q.Target) } if !sdk.IsKind(e, sdk.ErrSandboxNotFound) { return wire.State{}, e @@ -172,17 +256,6 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, if q.ObserveOnly || q.Target.ID != "" { return wire.State{}, wire.ErrUnconfirmed } - source, err := sdk.GetSandbox(ctx, q.Snapshot.SourceName) - if err == nil { - if source.ID() != q.Snapshot.SourceID { - return wire.State{}, sandbox.ErrOwnership - } - if !terminal(source.Status()) { - return wire.State{}, wire.ErrUnconfirmed - } - } else if !sdk.IsKind(err, sdk.ErrSandboxNotFound) { - return wire.State{}, err - } restore := sdk.RestoreConfig{NetworkPolicy: b.network(), ExternalMountPolicy: sdk.ExternalMountStrict} if b.q.Workspace != nil { restore.Volumes = map[string]sdk.MountConfig{"/environment": sdk.Mount.Bind(b.q.Workspace.Path, sdk.MountOptions{})} @@ -195,7 +268,7 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, target.ID = live.ID() // The same completion path handles a fresh target and a previous Restore // whose response or derived proof write was interrupted. - state, err := b.finishRestore(ctx, target) + state, err := b.completeRestore(ctx, q, target) detachErr := live.Detach(context.Background()) if err != nil { return state, err @@ -206,21 +279,6 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, return state, nil } -// Observation consumes an already verified artifact identity and never changes -// source state. Execution qualification belongs to capture and subsequent use. -func observeCapturedSnapshot(source wire.Compute, snapshot wire.SnapshotIdentity, readOwned func(wire.Compute) (wire.State, error)) (wire.State, error) { - state, err := readOwned(source) - if err != nil && !sdk.IsKind(err, sdk.ErrSandboxNotFound) { - return wire.State{}, err - } - stopped := sdk.IsKind(err, sdk.ErrSandboxNotFound) || terminal(sdk.SandboxStatus(state.Status)) - status := "suspended" - if !stopped { - status = state.Status - } - return wire.State{Compute: source, Status: status, BootstrapComplete: true, Snapshot: &snapshot, SourceStopped: stopped}, nil -} - func terminal(s sdk.SandboxStatus) bool { return s == sdk.SandboxStatusStopped || s == sdk.SandboxStatusCrashed } diff --git a/services/core/tools/microsandbox-provider/snapshot_observation_test.go b/services/core/tools/microsandbox-provider/snapshot_observation_test.go index cf2a80b07..94a9ac697 100644 --- a/services/core/tools/microsandbox-provider/snapshot_observation_test.go +++ b/services/core/tools/microsandbox-provider/snapshot_observation_test.go @@ -3,83 +3,260 @@ package main import ( - "encoding/json" "errors" + "fmt" + "os" + "path/filepath" "testing" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" wire "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" - sdk "github.com/superradcompany/microsandbox/sdk/go" ) -func TestCapturedSnapshotObservationDoesNotRequireExecutionQualification(t *testing.T) { - for _, drift := range []string{"cpu", "image", "missing snapshot proof", "foreign snapshot proof"} { - t.Run(drift, func(t *testing.T) { - config, actual := deploymentFixture() - ref := sandbox.Reference{TenantID: "tenant", EnvironmentID: "environment", AllocationID: "allocation"} - source := wire.Compute{Name: "original", ID: "local:original"} - labels := wire.Labels(config, ref) - labels[workspaceModeLabel] = "owned" - labels[bootstrapLabel] = "complete" - actual["labels"] = labels - snapshotLabels := map[string]string{workspaceModeLabel: "owned", resourceProofLabel: resourceProof(config)} - switch drift { - case "cpu": - actual["resources"].(map[string]any)["cpus"] = 1 - case "image": - actual["image"] = "runtime:drifted" - case "missing snapshot proof": - delete(snapshotLabels, resourceProofLabel) - case "foreign snapshot proof": - snapshotLabels[resourceProofLabel] = "foreign" +func TestRestoreObservationClosesOnlyNeverAdmittedOperation(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "exact-target", Generation: 1}, ObserveOnly: true} + closed, err := b.admitRestore(q) + if err != nil || !closed { + t.Fatal(closed, err) + } + q.ObserveOnly = false + // A late helper cannot replace the durable closed record with admission. + closed, err = b.admitRestore(q) + if err != nil || !closed { + t.Fatal("late restore admitted", closed, err) + } + q.OperationID = "77777777-7777-4777-8777-777777777777" + closed, err = b.admitRestore(q) + if err != nil || closed { + t.Fatal(closed, err) + } + // A crash after admission remains unknown, including when no target is yet + // visible. The helper never manufactures no-effect evidence from absence. + q.ObserveOnly = true + closed, err = b.admitRestore(q) + if err != nil || closed { + t.Fatal("admitted operation closed", closed, err) + } + q.ObserveOnly = false + if _, err = b.admitRestore(q); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("admitted operation replayed", err) + } + q.ObserveOnly = true + q.Target.Name = "foreign" + if _, err = b.admitRestore(q); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal("foreign target admitted", err) + } +} +func TestRestoreJournalRejectsPartialAndSymlinkRecords(t *testing.T) { + for _, kind := range []string{"partial", "symlink"} { + t.Run(kind, func(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) } - raw, _ := json.Marshal(actual) - if qualifyConfiguration(config, source, string(raw), false) == nil && qualifySnapshotResources(config, snapshotLabels) == nil { - t.Fatal("fixture did not disqualify execution") + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", ObserveOnly: true} + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + if kind == "partial" { + err = os.WriteFile(path, []byte("{"), 0600) + } else { + err = os.Symlink(filepath.Join(t.TempDir(), "missing"), path) } - // inspectSnapshot has already verified the exact artifact and closure. - snapshot := wire.SnapshotIdentity{ID: "verified", SourceID: source.ID, SourceName: source.Name} - reads := 0 - state, err := observeCapturedSnapshot(source, snapshot, func(want wire.Compute) (wire.State, error) { - reads++ - return qualifyCompute(config, ref, want, source.ID, "paused", string(raw)) - }) - if err != nil || state.Snapshot == nil || *state.Snapshot != snapshot || state.Compute != source || state.SourceStopped || reads != 1 { - t.Fatal("owned receipt blocked by execution drift", state, err, reads) + if err != nil { + t.Fatal(err) + } + if _, err = b.admitRestore(q); err == nil { + t.Fatal("unsafe record accepted") } }) } } +func TestArchiveDigestRejectsGuestReadableAndSymlinkArtifacts(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "archive") + if err := os.WriteFile(path, []byte("private RAM"), 0600); err != nil { + t.Fatal(err) + } + hash, size, err := archiveDigest(path) + if err != nil || len(hash) != 64 || size != 11 { + t.Fatal(hash, size, err) + } + if err = os.Chmod(path, 0644); err != nil { + t.Fatal(err) + } + if _, _, err = archiveDigest(path); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal(err) + } + link := filepath.Join(dir, "link") + if err = os.Symlink(path, link); err != nil { + t.Fatal(err) + } + if _, _, err = archiveDigest(link); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal(err) + } +} -func TestCapturedSnapshotObservationRetainsSourceOwnershipAndState(t *testing.T) { - source := wire.Compute{Name: "original", ID: "local:original"} - snapshot := wire.SnapshotIdentity{ID: "verified", SourceID: source.ID} - for _, test := range []struct { - name, status string - err error - stopped bool - }{ - {name: "running", status: "running"}, - {name: "unknown", status: "unknown"}, - {name: "stopped", status: "stopped", stopped: true}, - {name: "crashed", status: "crashed", stopped: true}, - {name: "absent", err: &sdk.Error{Kind: sdk.ErrSandboxNotFound}, stopped: true}, - {name: "foreign", err: sandbox.ErrOwnership}, - {name: "unavailable", err: wire.ErrUnconfirmed}, - } { - t.Run(test.name, func(t *testing.T) { - state, err := observeCapturedSnapshot(source, snapshot, func(wire.Compute) (wire.State, error) { - return wire.State{Compute: source, Status: test.status}, test.err - }) - if test.err != nil && !test.stopped { - if !errors.Is(err, test.err) || state.Snapshot != nil { - t.Fatal("unverified source returned a successful observation", state, err) +func TestAdmittedRestoreBlocksCleanupUntilExactCompletion(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "exact-target", Generation: 1}} + if _, err = b.admitRestore(q); err != nil { + t.Fatal(err) + } + if err = b.requireSettledRestores(&q.Target, nil); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("unknown target was releasable", err) + } + if err = b.requireSettledRestores(nil, &q.Snapshot); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("unknown artifact was releasable", err) + } + other := wire.Compute{Name: "other-source"} + if err = b.requireSettledRestores(&other, nil); err != nil { + t.Fatal("unrelated source cleanup blocked", err) + } + record := restoreAdmission{OperationID: q.OperationID, Target: q.Target, Snapshot: q.Snapshot, CompletedID: "local:exact"} + if err = durableJSON(filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json"), record); err != nil { + t.Fatal(err) + } + if err = b.requireSettledRestores(&q.Target, nil); err != nil || q.Target.ID != "local:exact" { + t.Fatal("cleanup did not pin completed incarnation", q.Target, err) + } + q.Target.ID = "local:foreign" + if err = b.requireSettledRestores(&q.Target, nil); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal("foreign cleanup accepted", err) + } +} + +func TestTargetCleanupFenceSurvivesReopenBeforeAndAfterAdmission(t *testing.T) { + for _, admitted := range []bool{false, true} { + t.Run(fmt.Sprint(admitted), func(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "exact-target", Generation: 1}} + if admitted { + if _, err = b.admitRestore(q); err != nil { + t.Fatal(err) } - return } - if err != nil || state.Snapshot == nil || state.SourceStopped != test.stopped { - t.Fatal("observation changed source termination evidence", state, err) + if err = durableJSON(filepath.Join(b.storeDirectory(), "k-"+q.Target.Name+".json"), q.Target); err != nil { + t.Fatal(err) + } + guard.Close() + guard, err = allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + if _, err = b.admitRestore(q); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("late restore crossed cleanup fence", err) + } + if admitted { + // Only the caller which has completed native Get/Kill/Remove/Get invokes + // this journal update. The live crash probe qualifies the native fence. + if err = b.recordRestoreCleanup(q.Target); err != nil { + t.Fatal(err) + } + if err = b.requireSettledRestores(nil, &q.Snapshot); err != nil { + t.Fatal("cleaned obligation retained", err) + } } }) } } + +func TestLostCaptureResponseStillExposesOwnedIdentityWhenArchiveIsDamaged(t *testing.T) { + for _, settled := range []bool{false, true} { + for _, damage := range []string{"corrupt", "missing"} { + t.Run(fmt.Sprintf("settled=%t/%s", settled, damage), func(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + source := request.Compute + source.ID = "local:known-source" + operation := "66666666-6666-4666-8666-666666666666" + compatibility, err := wire.CheckpointClass(request.Config) + if err != nil { + t.Fatal(err) + } + snapshot := wire.SnapshotIdentity{Compatibility: compatibility, Reference: wire.SnapshotReference(request.Config, request.Reference, operation), ID: "exact-snapshot", Digest: "digest", CheckpointID: "checkpoint", CheckpointRoot: "root", OperationID: operation, SourceName: source.Name, SourceID: source.ID} + dir := b.archiveDirectory(operation) + if err = privateDirectory(dir); err != nil { + t.Fatal(err) + } + archive := filepath.Join(dir, "archive.msb") + if err = os.WriteFile(archive, []byte("verified full archive"), 0600); err != nil { + t.Fatal(err) + } + hash, size, err := archiveDigest(archive) + if err != nil { + t.Fatal(err) + } + receipt := archiveReceipt{Snapshot: snapshot, SHA256: hash, Size: size, SourceTerminated: settled, SourceStopped: settled} + if err = durableJSON(filepath.Join(dir, "receipt.json"), receipt); err != nil { + t.Fatal(err) + } + if damage == "corrupt" { + err = os.WriteFile(archive, []byte("damaged"), 0600) + } else { + err = os.Remove(archive) + } + if err != nil { + t.Fatal(err) + } + // Core lost the capture response and therefore has no Snapshot yet. It + // needs this ownership identity to reach ordinary Kill/DeleteSnapshot. + state, err := b.suspend(t.Context(), wire.SuspendRequest{Reference: request.Reference, OperationID: operation, Source: source, ObserveOnly: true}) + if err != nil || state.Snapshot == nil || *state.Snapshot != snapshot || state.Compute != source || state.SourceStopped || state.Status != "unknown" || state.BootstrapComplete { + t.Fatalf("cleanup identity hidden or execution authorized: %+v %v", state, err) + } + if artifact, err := b.importArchive(t.Context(), snapshot); err == nil || artifact != nil { + t.Fatal("damaged archive was usable for restore", err) + } + if _, err = b.suspend(t.Context(), wire.SuspendRequest{Reference: request.Reference, OperationID: operation, Source: source}); err == nil { + t.Fatal("non-observation accepted unusable archive") + } + foreign := source + foreign.ID = "local:foreign" + denied, err := b.suspend(t.Context(), wire.SuspendRequest{Reference: request.Reference, OperationID: operation, Source: foreign, ObserveOnly: true}) + if !errors.Is(err, sandbox.ErrOwnership) || denied.Snapshot != nil { + t.Fatal("foreign request recovered owned identity", denied, err) + } + nativeRemoved := false + if err = b.deleteArchiveWithNative(*state.Snapshot, func() error { nativeRemoved = true; return nil }); err != nil { + t.Fatal("owned damaged archive could not be deleted", err) + } + if !nativeRemoved { + t.Fatal("native cleanup was skipped") + } + if _, err = os.Lstat(archive); !errors.Is(err, os.ErrNotExist) { + t.Fatal("archive remains", err) + } + deleted, err := b.receipt(operation) + if err != nil || !deleted.Deleted || deleted.Snapshot != snapshot { + t.Fatal("exact deletion fence missing", deleted, err) + } + }) + } + } +} From 0d9617d39b6bf2717742a91178857a3af012be29 Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Sat, 10 Oct 2026 18:27:51 +0000 Subject: [PATCH 2/6] Align checkpoint fixtures with explicit restore admission --- services/core/internal/sandbox/providers/generation_test.go | 2 +- .../core/tests/integration/runtime_node_generations_test.go | 6 +++++- .../integration/session_execution_configuration_test.go | 5 +++++ 3 files changed, 11 insertions(+), 2 deletions(-) diff --git a/services/core/internal/sandbox/providers/generation_test.go b/services/core/internal/sandbox/providers/generation_test.go index 0bacbe40d..60b42d297 100644 --- a/services/core/internal/sandbox/providers/generation_test.go +++ b/services/core/internal/sandbox/providers/generation_test.go @@ -21,7 +21,7 @@ func generationConfig(t *testing.T) (sandbox.NodeConfig, sandboxmicro.Native) { dir := t.TempDir() spec := validRegistrationSpec() spec.Resources.RootDiskMiB, spec.Resources.EnvironmentDiskMiB = 8192, 8192 - native := sandboxmicro.Native{HelperPath: filepath.Join(dir, "helper"), RuntimeHome: dir, RuntimePath: filepath.Join(dir, "msb"), FirmwarePath: filepath.Join(dir, "firmware"), + native := sandboxmicro.Native{HelperPath: filepath.Join(dir, "helper"), RuntimeHome: dir, CheckpointRoot: t.TempDir(), RuntimePath: filepath.Join(dir, "msb"), FirmwarePath: filepath.Join(dir, "firmware"), Network: sandboxmicro.Network{DefaultEgress: "allow", DefaultIngress: "deny"}} raw, err := json.Marshal(native) if err != nil { diff --git a/services/core/tests/integration/runtime_node_generations_test.go b/services/core/tests/integration/runtime_node_generations_test.go index 20e5779fb..3197a00c7 100644 --- a/services/core/tests/integration/runtime_node_generations_test.go +++ b/services/core/tests/integration/runtime_node_generations_test.go @@ -29,7 +29,11 @@ func changeNodeTarget(t *testing.T, w *Store, view deployment.View, input sandbo func generationHeartbeat(t *testing.T, s *Store, node deployment.Enrollment, connection string, view deployment.View, state string) { t.Helper() - err := deploymentService(t, s).HeartbeatGenerations(t.Context(), node.NodeID, connection, view.OwnerEpoch, deployment.NodeHealth{}, []sandbox.GenerationStatus{{Generation: view.Generation, SpecificationDigest: view.SpecificationDigest, State: state}}) + status := sandbox.GenerationStatus{Generation: view.Generation, SpecificationDigest: view.SpecificationDigest, State: state} + if state == "ready" && view.Provider == "microsandbox" { + status.Checkpoint = &sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"} + } + err := deploymentService(t, s).HeartbeatGenerations(t.Context(), node.NodeID, connection, view.OwnerEpoch, deployment.NodeHealth{}, []sandbox.GenerationStatus{status}) if err != nil { t.Fatal(err) } diff --git a/services/core/tests/integration/session_execution_configuration_test.go b/services/core/tests/integration/session_execution_configuration_test.go index e4a1c168c..3a047f2f8 100644 --- a/services/core/tests/integration/session_execution_configuration_test.go +++ b/services/core/tests/integration/session_execution_configuration_test.go @@ -272,6 +272,11 @@ func TestSessionExecutionConfigurationSurvivesSuspendResume(t *testing.T) { if phase == "running" { retained = nil } + if phase == "restoring" { + if err := deploymentService(t, s).TouchActivity(t.Context(), tenant, environment.ID); err != nil { + t.Fatal(err) + } + } owner = runtimeSuspensionStep(t, w, owner, phase, retained) before, err := deploymentStore(w).Activity(t.Context(), owner.ID) if err != nil { From 2deb8eea68df40c99df4292e1a6cd9605d698b4b Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Sat, 10 Oct 2026 18:51:01 +0000 Subject: [PATCH 3/6] Wake checkpoint destinations and settle canceled restore preparation --- docs/sandbox-provider.md | 4 +- docs/zh/sandbox-provider.md | 6 +- .../internal/execution/runtime_compute.go | 5 +- .../execution/runtime_compute_wake.go | 3 + .../internal/execution/runtime_lifecycle.go | 1 + .../internal/execution/runtime_manager.go | 15 +- .../execution/runtime_restore_attempt_test.go | 11 ++ .../execution/runtime_wake_hint_test.go | 29 +++ .../core/internal/sandbox/sandbox_provider.go | 2 +- .../tools/microsandbox-provider/README.md | 2 +- .../tools/microsandbox-provider/archive.go | 25 +++ .../restore_completion_test.go | 6 +- .../tools/microsandbox-provider/snapshot.go | 32 ++- .../snapshot_observation_test.go | 186 ++++++++++++++++++ 14 files changed, 312 insertions(+), 15 deletions(-) diff --git a/docs/sandbox-provider.md b/docs/sandbox-provider.md index d48ae3eeb..77bff895b 100644 --- a/docs/sandbox-provider.md +++ b/docs/sandbox-provider.md @@ -196,9 +196,9 @@ Queued work and live Environment file access wake a suspended Environment; histo A checkpoint-capable generation reports `CheckpointCompatibility`: opaque `ArtifactDomain` and `ExecutionClass` tokens. Every verified `SnapshotIdentity` carries the same qualification. Core compares these values without interpreting CPU features, filesystem paths or storage implementations. A restore destination must be online and ready for the allocation's immutable deployment generation, match both tokens, and have active capacity plus a retained slot when moving from another node. The allocation, Device and Session identities remain unchanged. The Session and deployment transaction commits the destination route, placement and exact restore intent before target-side native work. No capacity or compatible destination leaves the retained snapshot owned until its configured deadline. The target must already hold that exact generation; Core does not automatically prepare historical generations on new nodes. During upgrades, retain eligible nodes and their generation providers until their checkpoint retention obligations end. -`Suspend` may report `SourceStopped` only after publishing a verified full archive durably, stopping the exact source compute and settling source-local capture and cleanup obligations. Core persists the snapshot before marking it suspended; native absence alone is insufficient. `Suspend.ObserveOnly` may finish archive publication and source cleanup for an already verified snapshot of the same operation, but cannot capture another snapshot or stop a source that has resumed running. If a previously published archive is missing or damaged, `ObserveOnly` may return its exact durable ownership receipt with `Status: unknown` and `SourceStopped: false` for cleanup; that receipt proves neither current archive usability nor source absence. `Resume` still verifies the archive before execution. After the source-settlement barrier, source-node unavailability does not prevent restoration. Deletion and ordinary expiry can transfer an idle suspended allocation's cleanup route to an online node in the same artifact domain; deletion does not require execution-class compatibility. An unknown running or restoring writer never moves merely because its node is unreachable. +`Suspend` may report `SourceStopped` only after publishing a verified full archive durably, stopping the exact source compute and settling source-local capture and cleanup obligations. Core persists the snapshot before marking it suspended; native absence alone is insufficient. `Suspend.ObserveOnly` may finish archive publication and source cleanup for an already verified snapshot of the same operation, but cannot capture another snapshot or stop a source that has resumed running. If a previously published archive is missing or damaged, `ObserveOnly` may return its exact durable ownership receipt with `Status: unknown` and `SourceStopped: false` for cleanup; that receipt proves neither current archive usability nor source absence. `Resume` still verifies the archive before execution. After the source-settlement barrier, source-node unavailability does not prevent restoration. Deletion and ordinary expiry can transfer an idle suspended allocation's cleanup route to an online node in the same artifact domain; deletion does not require execution-class compatibility. The cleanup destination must already serve the same immutable generation and have retained capacity. Without such a node, cleanup stays pending and ownership is not released; operators must restore an eligible node or its capacity. An unknown running or restoring writer never moves merely because its node is unreachable. -Recovery uses `Resume.ObserveOnly` for the persisted target and operation. `RestoreAttemptClosed` is an exact operation ID, not a generic absence flag: the adapter may return it only after durably closing an attempt that never entered native execution and fencing every late request for that ID. Core then persists a new operation ID before attempting execution. An admitted attempt whose outcome is unknown retains the same target and ownership; a missing native listing, helper death or timeout does not authorize replay. User deletion still requires exact native settlement before resources and ownership can be released. +Recovery uses `Resume.ObserveOnly` for the persisted target and operation. `RestoreAttemptClosed` is an exact operation ID, not a generic absence flag: the adapter may return it only after durably closing an attempt that never entered native execution and fencing every late request for that ID. An adapter may also close a durably admitted attempt when it can prove that its current invocation failed before dispatching native restoration. The microsandbox adapter does this only for cancellation or deadline errors before native dispatch; archive corruption, qualification errors and opaque SDK failures do not authorize an automatic retry. Core then persists a new operation ID before attempting execution. Once native restoration has been dispatched, an unknown outcome retains the same target, ownership and original retention deadline; a missing native listing, helper death or timeout does not authorize replay. Such a restore can remain unavailable until explicit deletion or ordinary retention expiry, and pending inputs can reach their public deadline while waiting. User deletion still requires exact native settlement before resources and ownership can be released. The microsandbox adapter requires an explicit private `checkpoint_root`, outside every guest-accessible filesystem. Its directory is mode `0700`; a private namespace marker identifies the actual archive store. The default installation creates a node-local private store. Cross-node restoration requires operators to mount the same private store on each participating node; mount paths may differ. Runtime state directories and workspace bindings are never used to infer this root. The adapter verifies the archive, external filesystem identity and native execution class again before restoration. Restored RAM and file descriptors do not promise continuity of external TCP connections or replay safety for application side effects. diff --git a/docs/zh/sandbox-provider.md b/docs/zh/sandbox-provider.md index fc4a3eba2..90c570356 100644 --- a/docs/zh/sandbox-provider.md +++ b/docs/zh/sandbox-provider.md @@ -1,7 +1,7 @@ --- title: "添加 Sandbox Provider" source: docs/sandbox-provider.md -source_hash: 08e51f80e7b4a945bceacd98135c52f5f97d5ecb85bfc0b027b3fe9933ce7ce0 +source_hash: 85b04c5c6049144c49132fea14297cdacd3f3c1531c6c7f14f82156dd6cf1d80 --- **Sandbox Provider** 为 Core 管理的 Environment 提供 Runtime daemon 运行所需的外层计算资源,以及启动 daemon 的有界引导流程。本指南说明如何添加 Provider,并作为 Core 驱动 Provider 的参考。接口为 [`SandboxProvider`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/services/core/internal/sandbox/sandbox_provider.go)。 @@ -198,9 +198,9 @@ Worker lease、Session lock 与 per-node gate 对每个 provider 负责 suspensi 支持检查点的 generation 报告 `CheckpointCompatibility`,包含不透明的 `ArtifactDomain` 与 `ExecutionClass` token。每个已验证 `SnapshotIdentity` 携带同样的资格。Core 只比较这些值,不解释 CPU 特性、文件系统路径或存储实现。恢复目标必须在线、对 allocation 的不可变 deployment generation 已就绪、两个 token 均匹配,并有 active 容量;跨 node 时还需 retained slot。allocation、Device 和 Session 身份保持不变。Session 与 deployment 事务先提交目标路由、placement 和精确 restore intent,再执行目标侧原生操作。没有容量或兼容目标时,保留快照的所有权一直持续到配置期限。目标必须已持有该精确 generation;Core 不会在新节点自动准备历史 generation。升级期间应保留符合条件的节点及其 generation provider,直到检查点保留义务结束。 -`Suspend` 只有在持久发布已验证的完整归档、停止精确 source compute,并结清 source 本地 capture 与 cleanup 义务后,才能报告 `SourceStopped`。Core 先持久化快照,再标记 suspended;仅原生资源不存在并不足够。`Suspend.ObserveOnly` 可以为同一 operation 已验证的快照完成归档发布和源清理,但不能重新捕获快照,也不能停止已经恢复 running 的源。此前已发布的归档缺失或损坏时,`ObserveOnly` 可以返回其精确持久 ownership receipt,标记 `Status: unknown` 和 `SourceStopped: false`,供清理使用;该 receipt 不证明归档当前可用,也不证明源不存在。`Resume` 仍在执行前验证归档。完成源结清屏障后,source node 不可用不再阻止恢复。删除和正常到期可以将空闲 suspended allocation 的清理路由转给同一 artifact domain 内的在线 node;删除不要求 execution class 匹配。未知 running 或 restoring writer 不会仅因 node 不可达而迁移。 +`Suspend` 只有在持久发布已验证的完整归档、停止精确 source compute,并结清 source 本地 capture 与 cleanup 义务后,才能报告 `SourceStopped`。Core 先持久化快照,再标记 suspended;仅原生资源不存在并不足够。`Suspend.ObserveOnly` 可以为同一 operation 已验证的快照完成归档发布和源清理,但不能重新捕获快照,也不能停止已经恢复 running 的源。此前已发布的归档缺失或损坏时,`ObserveOnly` 可以返回其精确持久 ownership receipt,标记 `Status: unknown` 和 `SourceStopped: false`,供清理使用;该 receipt 不证明归档当前可用,也不证明源不存在。`Resume` 仍在执行前验证归档。完成源结清屏障后,source node 不可用不再阻止恢复。删除和正常到期可以将空闲 suspended allocation 的清理路由转给同一 artifact domain 内的在线 node;删除不要求 execution class 匹配。清理目标必须已提供同一不可变 generation,且有 retained 容量。没有这样的 node 时,清理保持 pending,不释放所有权;运维需恢复符合条件的 node 或其容量。未知 running 或 restoring writer 不会仅因 node 不可达而迁移。 -恢复用 `Resume.ObserveOnly` 观察已持久化的 target 与 operation。`RestoreAttemptClosed` 是精确 operation ID,不是通用 absence 标志:adapter 只有在持久关闭从未进入原生执行的 attempt,并隔离该 ID 的所有迟到请求后才能返回。Core 随后先持久化新的 operation ID,再尝试执行。已 admitted 但结果未知的 attempt 保留相同 target 与所有权;原生列表缺失、helper 死亡或超时都不授权重放。用户删除仍需精确原生 settlement,之后才能释放资源与所有权。 +恢复用 `Resume.ObserveOnly` 观察已持久化的 target 与 operation。`RestoreAttemptClosed` 是精确 operation ID,不是通用 absence 标志:adapter 只有在持久关闭从未进入原生执行的 attempt,并隔离该 ID 的所有迟到请求后才能返回。如果 adapter 能证明当前调用在派发原生恢复之前失败,也可以关闭已持久 admitted 的 attempt。microsandbox adapter 仅对原生派发之前的取消或 deadline 错误执行这种关闭;归档损坏、资格错误和不透明 SDK 失败都不授权自动重试。Core 随后先持久化新的 operation ID,再尝试执行。一旦已派发原生恢复,结果未知的 attempt 保留相同 target、所有权与原始 retention deadline;原生列表缺失、helper 死亡或超时都不授权重放。这种恢复可能一直不可用,直到显式删除或正常保留期到期;等待中的 input 可能达到其公共 deadline。用户删除仍需精确原生 settlement,之后才能释放资源与所有权。 microsandbox adapter 要求显式的私有 `checkpoint_root`,位于所有 guest 可访问文件系统之外。目录权限为 `0700`,私有 namespace marker 标识真实归档存储。默认安装创建 node 本地私有 store。跨 node 恢复要求运维为参与 node 挂载同一个私有 store,挂载路径可以不同。Runtime state 目录与 workspace binding 都不用于推导该 root。adapter 在恢复前再次验证归档、外部文件系统身份和原生 execution class。恢复 RAM 与文件描述符不保证外部 TCP 连接连续,也不保证应用副作用可安全重放。 diff --git a/services/core/internal/execution/runtime_compute.go b/services/core/internal/execution/runtime_compute.go index 3b97738e4..71567378d 100644 --- a/services/core/internal/execution/runtime_compute.go +++ b/services/core/internal/execution/runtime_compute.go @@ -241,6 +241,9 @@ func (r *runtimeLifecycle) restoreIdleCompute(ctx context.Context, p sandbox.San } if next.NodeID != r.nodeID { // The durable route and reservation belong to the target node's lane. + if r.hintDestination != nil { + r.hintDestination(next.NodeID) + } return nil } return r.restoreCompute(ctx, p, next, state, false) @@ -286,7 +289,7 @@ func (r *runtimeLifecycle) restoreCompute(ctx context.Context, p sandbox.Sandbox if !result.ClosesRestoreAttempt(sandbox.ResumeRequest{OperationID: state.RestoreID, Snapshot: *state.Snapshot, Target: *state.Target, ObserveOnly: observeOnly}) { return sandbox.ErrOwnership } - // Only an exact, durably closed never-admitted attempt can be replaced. + // Only an exact, durably closed attempt with no native dispatch can be replaced. state.RestoreID = uuid.NewString() next, err := r.saveCompute(ctx, owner, "restoring", state, owner.ComputeRetainedUntil) if err != nil { diff --git a/services/core/internal/execution/runtime_compute_wake.go b/services/core/internal/execution/runtime_compute_wake.go index bf16b5fb2..4f836dc6a 100644 --- a/services/core/internal/execution/runtime_compute_wake.go +++ b/services/core/internal/execution/runtime_compute_wake.go @@ -84,6 +84,9 @@ func (r *runtimeLifecycle) cleanupCompute(ctx context.Context, p sandbox.Sandbox return err } if next.NodeID != r.nodeID { + if r.hintDestination != nil { + r.hintDestination(next.NodeID) + } return nil } owner = next diff --git a/services/core/internal/execution/runtime_lifecycle.go b/services/core/internal/execution/runtime_lifecycle.go index 68e678d6e..a4a14cc11 100644 --- a/services/core/internal/execution/runtime_lifecycle.go +++ b/services/core/internal/execution/runtime_lifecycle.go @@ -63,6 +63,7 @@ type runtimeLifecycle struct { pendingCursor string connections map[string]*runtimeConnection wakeHints chan struct{} + hintDestination func(string) } func newRuntimeManager(owner Owner, deployments *deployment.Service, deploymentReader deployment.Reader, sessionReader sessions.Reader, registry *runtimegateway.Registry, config *RuntimeProvider) (*runtimeManager, error) { diff --git a/services/core/internal/execution/runtime_manager.go b/services/core/internal/execution/runtime_manager.go index 080a04547..5c3544e96 100644 --- a/services/core/internal/execution/runtime_manager.go +++ b/services/core/internal/execution/runtime_manager.go @@ -98,7 +98,7 @@ func (m *runtimeManager) node(id string) (*runtimeNode, error) { deployment: m.deployment, deployments: m.deploymentService, reader: m.deploymentReader, lease: m.lease, registry: m.registry, config: m.config, nodeID: id, gate: make(chan struct{}, 1), ctx: ctx, stop: stop, - connections: make(map[string]*runtimeConnection), wakeHints: make(chan struct{}, 1), + connections: make(map[string]*runtimeConnection), wakeHints: make(chan struct{}, 1), hintDestination: m.hintDestination, }} m.nodes[id] = n } @@ -118,6 +118,19 @@ func (m *runtimeManager) node(id string) (*runtimeNode, error) { return n, nil } +// A committed handoff can introduce a node before the next inventory scan. +// Reuse its ordinary lane and coalesced wake; polling recovers missed hints. +func (m *runtimeManager) hintDestination(id string) { + n, err := m.node(id) + if err != nil { + return + } + select { + case n.lifecycle.wakeHints <- struct{}{}: + default: + } +} + func (m *runtimeManager) hints(id string) chan<- struct{} { m.mu.Lock() defer m.mu.Unlock() diff --git a/services/core/internal/execution/runtime_restore_attempt_test.go b/services/core/internal/execution/runtime_restore_attempt_test.go index 304811277..c18833796 100644 --- a/services/core/internal/execution/runtime_restore_attempt_test.go +++ b/services/core/internal/execution/runtime_restore_attempt_test.go @@ -156,6 +156,14 @@ func TestCheckpointTransferRoutesBeforeTargetIO(t *testing.T) { if err != nil { t.Fatal(err) } + hints := 0 + r.hintDestination = func(id string) { + hints++ + committed, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || id != targetNode || committed.NodeID != id || committed.ID != owner.ID { + t.Fatal("hint preceded durable route or targeted wrong node", committed, err) + } + } if err = r.observe(t.Context(), source); err != nil { t.Fatal(err) } @@ -166,6 +174,9 @@ func TestCheckpointTransferRoutesBeforeTargetIO(t *testing.T) { if moved.NodeID != targetNode || moved.ID != owner.ID || moved.DeviceID != owner.DeviceID || provider.derived != 0 || len(provider.requests) != 0 || provider.kills != 0 || provider.snapshots != 0 { t.Fatal("source lane executed target effects or lost identity", moved) } + if hints != 1 { + t.Fatalf("expected one committed handoff hint, got %d", hints) + } // Simulate the independently scheduled target lane after the durable handoff. r.nodeID = targetNode err = r.observe(t.Context(), moved) diff --git a/services/core/internal/execution/runtime_wake_hint_test.go b/services/core/internal/execution/runtime_wake_hint_test.go index 955afc33c..006871999 100644 --- a/services/core/internal/execution/runtime_wake_hint_test.go +++ b/services/core/internal/execution/runtime_wake_hint_test.go @@ -8,6 +8,8 @@ import ( "sync/atomic" "testing" "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" ) type maintenanceTestLoop struct { @@ -355,3 +357,30 @@ func TestRuntimeMaintenanceCancellationAfterSuccessfulScanWinsReadyTriggers(t *t t.Fatalf("ready triggers bypassed cancellation: calls=%d, error=%v", calls, err) } } + +func TestDestinationHintCreatesAndReusesOrdinaryLane(t *testing.T) { + m := &runtimeManager{ctx: t.Context(), config: RuntimeProvider{Mode: string(sandbox.DeploymentNodes), Provider: &retentionProvider{}}, nodes: map[string]*runtimeNode{}} + m.hintDestination("target") + n := m.nodes["target"] + if n == nil || n.lifecycle.hintDestination == nil || len(n.lifecycle.wakeHints) != 1 { + t.Fatal("missing destination lane or hint") + } + for range 10 { + m.hintDestination("target") + } + if m.nodes["target"] != n || len(n.lifecycle.wakeHints) != 1 { + t.Fatal("hint duplicated lane or failed to coalesce") + } + m.switching = true + m.hintDestination("during-transition") + if m.nodes["during-transition"] != nil { + t.Fatal("hint bypassed configuration transition") + } + m.switching = false + m.closed = true + m.hintDestination("after-close") + if m.nodes["after-close"] != nil { + t.Fatal("hint bypassed closed manager") + } + n.lifecycle.stop() +} diff --git a/services/core/internal/sandbox/sandbox_provider.go b/services/core/internal/sandbox/sandbox_provider.go index e77a4c72d..568c16b07 100644 --- a/services/core/internal/sandbox/sandbox_provider.go +++ b/services/core/internal/sandbox/sandbox_provider.go @@ -178,7 +178,7 @@ type ComputeState struct { // and capture obligations have settled, so another node can own restoration. SourceStopped bool // RestoreAttemptClosed echoes the exact Resume OperationID only after a - // never-admitted attempt is durably closed against late execution. + // never-dispatched attempt is durably closed against late native execution. RestoreAttemptClosed string } diff --git a/services/core/tools/microsandbox-provider/README.md b/services/core/tools/microsandbox-provider/README.md index 0f69b881a..3e885bc70 100644 --- a/services/core/tools/microsandbox-provider/README.md +++ b/services/core/tools/microsandbox-provider/README.md @@ -40,7 +40,7 @@ Core persists operation IDs, source and target generations, exact identities and After a lost response Core uses `ObserveOnly`. It never starts a capture or restore. For an existing verified capture it may complete archive publication and exact paused-source cleanup, but never captures again or kills a running source. Source termination is recorded before local removal so a lost cleanup response can be settled from the archive receipt. A valid ownership receipt remains discoverable when archive bytes are missing or corrupt: observation returns its snapshot identity with unknown source state and `SourceStopped=false`, so ordinary explicit cleanup can proceed. Restore still requires complete archive verification. An interrupted restore may finish the derived proof on the exact running target, but never restarts a stopped target or restores again. -Before native import or restore, the helper synchronizes an admission record bound to the request operation, snapshot and target. Observation of a never-admitted operation writes a permanent closed record and returns that exact `RestoreAttemptClosed`; late dispatch is rejected. After admission, helper death or native absence alone never establishes no effect. Successful resume requires an observable exact running target. A failed or uncertain admitted restore retains its artifact and owner until explicit deletion or the original retention deadline; the normal cleanup operation uses the native lifecycle fence and records completion. It never shortens retention or substitutes a cold restart. +Before native import or restore, the helper synchronizes an admission record bound to the request operation, snapshot and target. Observation of a never-admitted operation writes a permanent closed record and returns that exact `RestoreAttemptClosed`; late dispatch is rejected. A fresh dispatch explicitly canceled or deadline-exceeded during archive import or qualification before calling native Restore may also durably close its exact attempt while still holding the allocation lock. Permanent or opaque errors keep the admission open, avoiding repeated attempts against an invalid archive. This proof is never inferred from a previous admission or a native Restore error. After admission, helper death or native absence alone never establishes no effect. Observation checks an admitted target before reading the archive; typed absence or a non-running target remains unconfirmed without hashing or importing the archive. A running target still requires complete archive verification. Successful resume requires an observable exact running target. A failed or uncertain admitted restore retains its artifact and owner until explicit deletion or the original retention deadline; the normal cleanup operation uses the native lifecycle fence and records completion. It never shortens retention or substitutes a cold restart. GetCompute, commands, cleanup and the next suspension verify restored provenance from the persisted VM configuration after the consumed artifact is deleted. diff --git a/services/core/tools/microsandbox-provider/archive.go b/services/core/tools/microsandbox-provider/archive.go index f3d159759..c9f76a48c 100644 --- a/services/core/tools/microsandbox-provider/archive.go +++ b/services/core/tools/microsandbox-provider/archive.go @@ -330,6 +330,31 @@ func (b backend) admitRestore(q wire.ResumeRequest) (bool, error) { return want.Closed, nil } +// Only the current fresh dispatch, under the allocation lock and before its +// first RestoreSandbox call, may record this proof. A recovered open admission +// or native restore error cannot establish that no guest execution occurred. +func (b backend) closeUndispatchedRestore(q wire.ResumeRequest, cause error) error { + // Only explicit cancellation/deadline errors warrant another dispatch. + // Permanent or opaque failures retain this admission and its original expiry. + if !errors.Is(cause, context.Canceled) && !errors.Is(cause, context.DeadlineExceeded) { + return nil + } + if q.ObserveOnly || q.Target.ID != "" { + return sandbox.ErrOwnership + } + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + var prior restoreAdmission + if err := readPrivateJSON(path, &prior); err != nil { + return err + } + want := restoreAdmission{OperationID: q.OperationID, Snapshot: q.Snapshot, Target: q.Target} + if !reflect.DeepEqual(prior, want) { + return sandbox.ErrOwnership + } + want.Closed = true + return durableJSON(path, want) +} + // Completion records the incarnation whose native restore has become observable // and whose full provenance and running state have been verified. This is // execution evidence; cleanup separately fences dispatch and uses native locks. diff --git a/services/core/tools/microsandbox-provider/restore_completion_test.go b/services/core/tools/microsandbox-provider/restore_completion_test.go index 2dbd5ec55..5f0b13a14 100644 --- a/services/core/tools/microsandbox-provider/restore_completion_test.go +++ b/services/core/tools/microsandbox-provider/restore_completion_test.go @@ -61,7 +61,7 @@ func TestRestoreCompletionRecoversOnlyDerivedProof(t *testing.T) { } func TestRestoreCompletionRejectsUnverifiedTargets(t *testing.T) { - for _, fault := range []string{"cpu", "memory", "environment", "image", "root", "foreign parent", "foreign ID", "foreign proof", "stopped", "unfinished", "replacement after write", "proof not persisted"} { + for _, fault := range []string{"cpu", "memory", "environment", "image", "root", "foreign parent", "foreign ID", "foreign proof", "stopped", "starting", "paused", "crashed", "unfinished", "replacement after write", "proof not persisted"} { t.Run(fault, func(t *testing.T) { config, actual := deploymentFixture() actual["image"].(map[string]any)["Oci"].(map[string]any)["root_disk"].(map[string]any)["size_mib"] = nil @@ -87,8 +87,8 @@ func TestRestoreCompletionRejectsUnverifiedTargets(t *testing.T) { actualID = "local:foreign" case "foreign proof": labels[resourceProofLabel] = "foreign" - case "stopped": - status = "stopped" + case "stopped", "starting", "paused", "crashed": + status = fault case "unfinished": actual["checkpoint_restore"] = map[string]string{"checkpoint_id": "pending"} } diff --git a/services/core/tools/microsandbox-provider/snapshot.go b/services/core/tools/microsandbox-provider/snapshot.go index 6d5ebb79a..56503d69d 100644 --- a/services/core/tools/microsandbox-provider/snapshot.go +++ b/services/core/tools/microsandbox-provider/snapshot.go @@ -234,17 +234,43 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, target.ID = "" return wire.State{Compute: target, Status: "absent", RestoreAttemptClosed: q.OperationID}, nil } + // An existing admitted observation may cheaply remain unknown. Native absence + // cannot close it: a detached launcher could still make the target visible. + if q.ObserveOnly { + _, observed, err := b.inspectOwned(ctx, q.Target) + if sdk.IsKind(err, sdk.ErrSandboxNotFound) { + return wire.State{}, wire.ErrUnconfirmed + } + if err != nil { + return wire.State{}, err + } + if observed.Status != "running" { + return wire.State{}, wire.ErrUnconfirmed + } + } + // admitRestore rejects non-observation requests for every existing open + // record. Thus only this fresh dispatch, still holding the allocation lock, + // can certify a failure before RestoreSandbox was called. Never use this + // closure for a native restore error or for a recovered admission. + beforeDispatchFailure := func(cause error) (wire.State, error) { + if !q.ObserveOnly { + if err := b.closeUndispatchedRestore(q, cause); err != nil { + return wire.State{}, errors.Join(cause, err) + } + } + return wire.State{}, cause + } // Admission is durable before any native import/restore operation. A prior // admitted operation with no visible target remains unknown, never replayed. artifact, e := b.importArchive(ctx, q.Snapshot) if e != nil { - return wire.State{}, e + return beforeDispatchFailure(e) } if e = qualifySnapshotResources(b.q.Config, artifact.Labels()); e != nil { - return wire.State{}, e + return beforeDispatchFailure(e) } if err := qualifyWorkspace(artifact.Labels(), b.q.Workspace); err != nil { - return wire.State{}, err + return beforeDispatchFailure(err) } _, _, e = b.inspectOwned(ctx, q.Target) if e == nil { diff --git a/services/core/tools/microsandbox-provider/snapshot_observation_test.go b/services/core/tools/microsandbox-provider/snapshot_observation_test.go index 94a9ac697..a6e6b210d 100644 --- a/services/core/tools/microsandbox-provider/snapshot_observation_test.go +++ b/services/core/tools/microsandbox-provider/snapshot_observation_test.go @@ -3,6 +3,7 @@ package main import ( + "context" "errors" "fmt" "os" @@ -260,3 +261,188 @@ func TestLostCaptureResponseStillExposesOwnedIdentityWhenArchiveIsDamaged(t *tes } } } + +// These requests use the real SDK's empty local registry, without starting a VM. +func TestPermanentRestoreImportFailureKeepsOneAdmission(t *testing.T) { + for _, damage := range []string{"missing", "corrupt"} { + t.Run(damage, func(t *testing.T) { + request := initialInfoRequest(t) + t.Setenv("MSB_HOME", request.Config.RuntimeHome) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: wire.Name(request.Config, request.Reference, 1), Generation: 1}} + if damage == "corrupt" { + compatibility, err := wire.CheckpointClass(request.Config) + if err != nil { + t.Fatal(err) + } + q.Snapshot = wire.SnapshotIdentity{Compatibility: compatibility, Reference: wire.SnapshotReference(request.Config, request.Reference, q.OperationID), ID: "id", Digest: "digest", CheckpointID: "checkpoint", CheckpointRoot: "root", OperationID: q.OperationID, SourceName: wire.Name(request.Config, request.Reference, 0), SourceID: "local:source"} + dir := b.archiveDirectory(q.OperationID) + if err = privateDirectory(dir); err != nil { + t.Fatal(err) + } + archive := filepath.Join(dir, "archive.msb") + if err = os.WriteFile(archive, []byte("original"), 0600); err != nil { + t.Fatal(err) + } + hash, size, err := archiveDigest(archive) + if err != nil { + t.Fatal(err) + } + if err = durableJSON(filepath.Join(dir, "receipt.json"), archiveReceipt{Snapshot: q.Snapshot, SHA256: hash, Size: size, SourceStopped: true, SourceTerminated: true}); err != nil { + t.Fatal(err) + } + if err = os.WriteFile(archive, []byte("corrupt!"), 0600); err != nil { + t.Fatal(err) + } + } + if _, err = b.resume(t.Context(), q); err == nil { + t.Fatal("missing archive accepted") + } + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + var record restoreAdmission + if err = readPrivateJSON(path, &record); err != nil || record.Closed { + t.Fatal("permanent failure closed", record, err) + } + before, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + files, err := os.ReadDir(b.storeDirectory()) + if err != nil { + t.Fatal(err) + } + q.ObserveOnly = true + for i := 0; i < 3; i++ { + state, err := b.resume(t.Context(), q) + if !errors.Is(err, wire.ErrUnconfirmed) || state.RestoreAttemptClosed != "" { + t.Fatal(state, err) + } + } + after, err := os.ReadFile(path) + if err != nil || string(before) != string(after) { + t.Fatal("journal changed", err) + } + afterFiles, err := os.ReadDir(b.storeDirectory()) + if err != nil || len(afterFiles) != len(files) { + t.Fatal("journal grew", err) + } + }) + } +} + +func TestUndispatchedFailureClosesOnlyExplicitCancellation(t *testing.T) { + for _, cause := range []error{context.Canceled, fmt.Errorf("wrapped: %w", context.DeadlineExceeded), sandbox.ErrOwnership, sandbox.ErrInvalid, os.ErrNotExist, errors.New("opaque SDK failure")} { + t.Run(cause.Error(), func(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "exact-target", Generation: 1}} + if _, err = b.admitRestore(q); err != nil { + t.Fatal(err) + } + if err = b.closeUndispatchedRestore(q, cause); err != nil { + t.Fatal(err) + } + guard.Close() + guard, err = allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + q.ObserveOnly = true + closed, err := b.admitRestore(q) + want := errors.Is(cause, context.Canceled) || errors.Is(cause, context.DeadlineExceeded) + if err != nil || closed != want { + t.Fatal("wrong closure", closed, want, err) + } + q.ObserveOnly = false + closed, err = b.admitRestore(q) + if want { + if err != nil || !closed { + t.Fatal("late dispatch permitted", closed, err) + } + } else if !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("old admission replayed", err) + } + }) + } +} + +func TestAdmittedAbsentObservationSkipsArchiveAndStaysUnknown(t *testing.T) { + request := initialInfoRequest(t) + t.Setenv("MSB_HOME", request.Config.RuntimeHome) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: wire.Name(request.Config, request.Reference, 1), Generation: 1}} + if _, err = b.admitRestore(q); err != nil { + t.Fatal(err) + } + // Even a missing archive must not be consulted: its ENOENT would differ from + // the typed unknown result returned for this existing admitted target. + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + before, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + q.ObserveOnly = true + state, err := b.resume(t.Context(), q) + if !errors.Is(err, wire.ErrUnconfirmed) || state.RestoreAttemptClosed != "" { + t.Fatal(state, err) + } + after, err := os.ReadFile(path) + if err != nil || string(before) != string(after) { + t.Fatal("observation mutated admission", err) + } + q.ObserveOnly = false + if _, err = b.resume(t.Context(), q); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("old admission replayed", err) + } +} + +func TestPreDispatchClosureRejectsConflictingJournal(t *testing.T) { + for _, kind := range []string{"observe", "completed", "cleaned", "foreign", "missing"} { + t.Run(kind, func(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "target", Generation: 1}} + r := restoreAdmission{OperationID: q.OperationID, Target: q.Target, Snapshot: q.Snapshot} + switch kind { + case "observe": + q.ObserveOnly = true + case "completed": + r.CompletedID = "local:1" + case "cleaned": + r.Cleaned = true + case "foreign": + r.Target.Name = "foreign" + } + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + if kind != "missing" { + if err = durableJSON(path, r); err != nil { + t.Fatal(err) + } + } + if err = b.closeUndispatchedRestore(q, context.Canceled); err == nil { + t.Fatal("unsafe closure accepted") + } + }) + } +} From b52b193f8e7effe44ce7a240fda9523e2dcfc57d Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Sat, 10 Oct 2026 18:58:48 +0000 Subject: [PATCH 4/6] Close expired restore requests before native dispatch --- docs/sandbox-provider.md | 2 +- docs/zh/sandbox-provider.md | 4 +- .../tools/microsandbox-provider/README.md | 2 +- .../tools/microsandbox-provider/archive.go | 12 +- .../tools/microsandbox-provider/snapshot.go | 25 ++- .../snapshot_observation_test.go | 149 +++++++++++++----- 6 files changed, 130 insertions(+), 64 deletions(-) diff --git a/docs/sandbox-provider.md b/docs/sandbox-provider.md index 77bff895b..f70b826dc 100644 --- a/docs/sandbox-provider.md +++ b/docs/sandbox-provider.md @@ -198,7 +198,7 @@ A checkpoint-capable generation reports `CheckpointCompatibility`: opaque `Artif `Suspend` may report `SourceStopped` only after publishing a verified full archive durably, stopping the exact source compute and settling source-local capture and cleanup obligations. Core persists the snapshot before marking it suspended; native absence alone is insufficient. `Suspend.ObserveOnly` may finish archive publication and source cleanup for an already verified snapshot of the same operation, but cannot capture another snapshot or stop a source that has resumed running. If a previously published archive is missing or damaged, `ObserveOnly` may return its exact durable ownership receipt with `Status: unknown` and `SourceStopped: false` for cleanup; that receipt proves neither current archive usability nor source absence. `Resume` still verifies the archive before execution. After the source-settlement barrier, source-node unavailability does not prevent restoration. Deletion and ordinary expiry can transfer an idle suspended allocation's cleanup route to an online node in the same artifact domain; deletion does not require execution-class compatibility. The cleanup destination must already serve the same immutable generation and have retained capacity. Without such a node, cleanup stays pending and ownership is not released; operators must restore an eligible node or its capacity. An unknown running or restoring writer never moves merely because its node is unreachable. -Recovery uses `Resume.ObserveOnly` for the persisted target and operation. `RestoreAttemptClosed` is an exact operation ID, not a generic absence flag: the adapter may return it only after durably closing an attempt that never entered native execution and fencing every late request for that ID. An adapter may also close a durably admitted attempt when it can prove that its current invocation failed before dispatching native restoration. The microsandbox adapter does this only for cancellation or deadline errors before native dispatch; archive corruption, qualification errors and opaque SDK failures do not authorize an automatic retry. Core then persists a new operation ID before attempting execution. Once native restoration has been dispatched, an unknown outcome retains the same target, ownership and original retention deadline; a missing native listing, helper death or timeout does not authorize replay. Such a restore can remain unavailable until explicit deletion or ordinary retention expiry, and pending inputs can reach their public deadline while waiting. User deletion still requires exact native settlement before resources and ownership can be released. +Recovery uses `Resume.ObserveOnly` for the persisted target and operation. `RestoreAttemptClosed` is an exact operation ID, not a generic absence flag: the adapter may return it only after durably closing an attempt that never entered native execution and fencing every late request for that ID. An adapter may also close a durably admitted attempt when it can prove that its current invocation failed before dispatching native restoration. The microsandbox adapter does this only when the fresh request's wire deadline expires before native dispatch; archive corruption, qualification errors and opaque SDK failures do not authorize an automatic retry. Core then persists a new operation ID before attempting execution. Once native restoration has been dispatched, an unknown outcome retains the same target, ownership and original retention deadline; a missing native listing, helper death or timeout does not authorize replay. Such a restore can remain unavailable until explicit deletion or ordinary retention expiry, and pending inputs can reach their public deadline while waiting. User deletion still requires exact native settlement before resources and ownership can be released. The microsandbox adapter requires an explicit private `checkpoint_root`, outside every guest-accessible filesystem. Its directory is mode `0700`; a private namespace marker identifies the actual archive store. The default installation creates a node-local private store. Cross-node restoration requires operators to mount the same private store on each participating node; mount paths may differ. Runtime state directories and workspace bindings are never used to infer this root. The adapter verifies the archive, external filesystem identity and native execution class again before restoration. Restored RAM and file descriptors do not promise continuity of external TCP connections or replay safety for application side effects. diff --git a/docs/zh/sandbox-provider.md b/docs/zh/sandbox-provider.md index 90c570356..ff6501786 100644 --- a/docs/zh/sandbox-provider.md +++ b/docs/zh/sandbox-provider.md @@ -1,7 +1,7 @@ --- title: "添加 Sandbox Provider" source: docs/sandbox-provider.md -source_hash: 85b04c5c6049144c49132fea14297cdacd3f3c1531c6c7f14f82156dd6cf1d80 +source_hash: 63ad2e39e7536caee9bd43f9b643073d32b32147f1440d188d6d38752628adc7 --- **Sandbox Provider** 为 Core 管理的 Environment 提供 Runtime daemon 运行所需的外层计算资源,以及启动 daemon 的有界引导流程。本指南说明如何添加 Provider,并作为 Core 驱动 Provider 的参考。接口为 [`SandboxProvider`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/services/core/internal/sandbox/sandbox_provider.go)。 @@ -200,7 +200,7 @@ Worker lease、Session lock 与 per-node gate 对每个 provider 负责 suspensi `Suspend` 只有在持久发布已验证的完整归档、停止精确 source compute,并结清 source 本地 capture 与 cleanup 义务后,才能报告 `SourceStopped`。Core 先持久化快照,再标记 suspended;仅原生资源不存在并不足够。`Suspend.ObserveOnly` 可以为同一 operation 已验证的快照完成归档发布和源清理,但不能重新捕获快照,也不能停止已经恢复 running 的源。此前已发布的归档缺失或损坏时,`ObserveOnly` 可以返回其精确持久 ownership receipt,标记 `Status: unknown` 和 `SourceStopped: false`,供清理使用;该 receipt 不证明归档当前可用,也不证明源不存在。`Resume` 仍在执行前验证归档。完成源结清屏障后,source node 不可用不再阻止恢复。删除和正常到期可以将空闲 suspended allocation 的清理路由转给同一 artifact domain 内的在线 node;删除不要求 execution class 匹配。清理目标必须已提供同一不可变 generation,且有 retained 容量。没有这样的 node 时,清理保持 pending,不释放所有权;运维需恢复符合条件的 node 或其容量。未知 running 或 restoring writer 不会仅因 node 不可达而迁移。 -恢复用 `Resume.ObserveOnly` 观察已持久化的 target 与 operation。`RestoreAttemptClosed` 是精确 operation ID,不是通用 absence 标志:adapter 只有在持久关闭从未进入原生执行的 attempt,并隔离该 ID 的所有迟到请求后才能返回。如果 adapter 能证明当前调用在派发原生恢复之前失败,也可以关闭已持久 admitted 的 attempt。microsandbox adapter 仅对原生派发之前的取消或 deadline 错误执行这种关闭;归档损坏、资格错误和不透明 SDK 失败都不授权自动重试。Core 随后先持久化新的 operation ID,再尝试执行。一旦已派发原生恢复,结果未知的 attempt 保留相同 target、所有权与原始 retention deadline;原生列表缺失、helper 死亡或超时都不授权重放。这种恢复可能一直不可用,直到显式删除或正常保留期到期;等待中的 input 可能达到其公共 deadline。用户删除仍需精确原生 settlement,之后才能释放资源与所有权。 +恢复用 `Resume.ObserveOnly` 观察已持久化的 target 与 operation。`RestoreAttemptClosed` 是精确 operation ID,不是通用 absence 标志:adapter 只有在持久关闭从未进入原生执行的 attempt,并隔离该 ID 的所有迟到请求后才能返回。如果 adapter 能证明当前调用在派发原生恢复之前失败,也可以关闭已持久 admitted 的 attempt。microsandbox adapter 仅在 fresh 请求的 wire deadline 于原生派发之前到期时执行这种关闭;归档损坏、资格错误和不透明 SDK 失败都不授权自动重试。Core 随后先持久化新的 operation ID,再尝试执行。一旦已派发原生恢复,结果未知的 attempt 保留相同 target、所有权与原始 retention deadline;原生列表缺失、helper 死亡或超时都不授权重放。这种恢复可能一直不可用,直到显式删除或正常保留期到期;等待中的 input 可能达到其公共 deadline。用户删除仍需精确原生 settlement,之后才能释放资源与所有权。 microsandbox adapter 要求显式的私有 `checkpoint_root`,位于所有 guest 可访问文件系统之外。目录权限为 `0700`,私有 namespace marker 标识真实归档存储。默认安装创建 node 本地私有 store。跨 node 恢复要求运维为参与 node 挂载同一个私有 store,挂载路径可以不同。Runtime state 目录与 workspace binding 都不用于推导该 root。adapter 在恢复前再次验证归档、外部文件系统身份和原生 execution class。恢复 RAM 与文件描述符不保证外部 TCP 连接连续,也不保证应用副作用可安全重放。 diff --git a/services/core/tools/microsandbox-provider/README.md b/services/core/tools/microsandbox-provider/README.md index 3e885bc70..f30db5987 100644 --- a/services/core/tools/microsandbox-provider/README.md +++ b/services/core/tools/microsandbox-provider/README.md @@ -40,7 +40,7 @@ Core persists operation IDs, source and target generations, exact identities and After a lost response Core uses `ObserveOnly`. It never starts a capture or restore. For an existing verified capture it may complete archive publication and exact paused-source cleanup, but never captures again or kills a running source. Source termination is recorded before local removal so a lost cleanup response can be settled from the archive receipt. A valid ownership receipt remains discoverable when archive bytes are missing or corrupt: observation returns its snapshot identity with unknown source state and `SourceStopped=false`, so ordinary explicit cleanup can proceed. Restore still requires complete archive verification. An interrupted restore may finish the derived proof on the exact running target, but never restarts a stopped target or restores again. -Before native import or restore, the helper synchronizes an admission record bound to the request operation, snapshot and target. Observation of a never-admitted operation writes a permanent closed record and returns that exact `RestoreAttemptClosed`; late dispatch is rejected. A fresh dispatch explicitly canceled or deadline-exceeded during archive import or qualification before calling native Restore may also durably close its exact attempt while still holding the allocation lock. Permanent or opaque errors keep the admission open, avoiding repeated attempts against an invalid archive. This proof is never inferred from a previous admission or a native Restore error. After admission, helper death or native absence alone never establishes no effect. Observation checks an admitted target before reading the archive; typed absence or a non-running target remains unconfirmed without hashing or importing the archive. A running target still requires complete archive verification. Successful resume requires an observable exact running target. A failed or uncertain admitted restore retains its artifact and owner until explicit deletion or the original retention deadline; the normal cleanup operation uses the native lifecycle fence and records completion. It never shortens retention or substitutes a cold restart. +Before native import or restore, the helper synchronizes an admission record bound to the request operation, snapshot and target. Observation of a never-admitted operation writes a permanent closed record and returns that exact `RestoreAttemptClosed`; late dispatch is rejected. A fresh dispatch checks its existing request deadline after admission and immediately before calling native Restore. If the deadline has elapsed, it durably closes its exact attempt while still holding the allocation lock and returns without invoking Restore. Native FFI waits remain uncancelled. Import or qualification errors keep the admission open, avoiding repeated attempts against an invalid archive. This proof is never inferred from a previous admission or a native Restore error. After admission, helper death or native absence alone never establishes no effect. Observation checks an admitted target before reading the archive; typed absence or a non-running target remains unconfirmed without hashing or importing the archive. A running target still requires complete archive verification. Successful resume requires an observable exact running target. A failed or uncertain admitted restore retains its artifact and owner until explicit deletion or the original retention deadline; the normal cleanup operation uses the native lifecycle fence and records completion. It never shortens retention or substitutes a cold restart. GetCompute, commands, cleanup and the next suspension verify restored provenance from the persisted VM configuration after the consumed artifact is deleted. diff --git a/services/core/tools/microsandbox-provider/archive.go b/services/core/tools/microsandbox-provider/archive.go index c9f76a48c..608a5bdf7 100644 --- a/services/core/tools/microsandbox-provider/archive.go +++ b/services/core/tools/microsandbox-provider/archive.go @@ -13,6 +13,7 @@ import ( "path/filepath" "reflect" "strings" + "time" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" wire "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" @@ -333,10 +334,8 @@ func (b backend) admitRestore(q wire.ResumeRequest) (bool, error) { // Only the current fresh dispatch, under the allocation lock and before its // first RestoreSandbox call, may record this proof. A recovered open admission // or native restore error cannot establish that no guest execution occurred. -func (b backend) closeUndispatchedRestore(q wire.ResumeRequest, cause error) error { - // Only explicit cancellation/deadline errors warrant another dispatch. - // Permanent or opaque failures retain this admission and its original expiry. - if !errors.Is(cause, context.Canceled) && !errors.Is(cause, context.DeadlineExceeded) { +func (b backend) closeExpiredRestore(q wire.ResumeRequest) error { + if time.Now().Before(b.q.Deadline) { return nil } if q.ObserveOnly || q.Target.ID != "" { @@ -352,7 +351,10 @@ func (b backend) closeUndispatchedRestore(q wire.ResumeRequest, cause error) err return sandbox.ErrOwnership } want.Closed = true - return durableJSON(path, want) + if err := durableJSON(path, want); err != nil { + return err + } + return context.DeadlineExceeded } // Completion records the incarnation whose native restore has become observable diff --git a/services/core/tools/microsandbox-provider/snapshot.go b/services/core/tools/microsandbox-provider/snapshot.go index 56503d69d..01475786b 100644 --- a/services/core/tools/microsandbox-provider/snapshot.go +++ b/services/core/tools/microsandbox-provider/snapshot.go @@ -248,29 +248,25 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, return wire.State{}, wire.ErrUnconfirmed } } - // admitRestore rejects non-observation requests for every existing open - // record. Thus only this fresh dispatch, still holding the allocation lock, - // can certify a failure before RestoreSandbox was called. Never use this - // closure for a native restore error or for a recovered admission. - beforeDispatchFailure := func(cause error) (wire.State, error) { - if !q.ObserveOnly { - if err := b.closeUndispatchedRestore(q, cause); err != nil { - return wire.State{}, errors.Join(cause, err) - } + // Only a fresh dispatch may close its attempt before native execution. + // Lifecycle FFI calls use an uncancelled context; the request deadline is + // checked between settled operations, never used to cancel a native wait. + if !q.ObserveOnly { + if err := b.closeExpiredRestore(q); err != nil { + return wire.State{}, err } - return wire.State{}, cause } // Admission is durable before any native import/restore operation. A prior // admitted operation with no visible target remains unknown, never replayed. artifact, e := b.importArchive(ctx, q.Snapshot) if e != nil { - return beforeDispatchFailure(e) + return wire.State{}, e } if e = qualifySnapshotResources(b.q.Config, artifact.Labels()); e != nil { - return beforeDispatchFailure(e) + return wire.State{}, e } if err := qualifyWorkspace(artifact.Labels(), b.q.Workspace); err != nil { - return beforeDispatchFailure(err) + return wire.State{}, err } _, _, e = b.inspectOwned(ctx, q.Target) if e == nil { @@ -286,6 +282,9 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, if b.q.Workspace != nil { restore.Volumes = map[string]sdk.MountConfig{"/environment": sdk.Mount.Bind(b.q.Workspace.Path, sdk.MountOptions{})} } + if err := b.closeExpiredRestore(q); err != nil { + return wire.State{}, err + } live, e := sdk.RestoreSandbox(ctx, artifact, q.Target.Name, sdk.WithRestoreConfig(restore)) if e != nil { return wire.State{}, e diff --git a/services/core/tools/microsandbox-provider/snapshot_observation_test.go b/services/core/tools/microsandbox-provider/snapshot_observation_test.go index a6e6b210d..74810cc67 100644 --- a/services/core/tools/microsandbox-provider/snapshot_observation_test.go +++ b/services/core/tools/microsandbox-provider/snapshot_observation_test.go @@ -7,8 +7,10 @@ import ( "errors" "fmt" "os" + "os/exec" "path/filepath" "testing" + "time" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" wire "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" @@ -264,10 +266,15 @@ func TestLostCaptureResponseStillExposesOwnedIdentityWhenArchiveIsDamaged(t *tes // These requests use the real SDK's empty local registry, without starting a VM. func TestPermanentRestoreImportFailureKeepsOneAdmission(t *testing.T) { + if !isolatedSDKTest(t) { + return + } + home := t.TempDir() + t.Setenv("MSB_HOME", home) for _, damage := range []string{"missing", "corrupt"} { t.Run(damage, func(t *testing.T) { request := initialInfoRequest(t) - t.Setenv("MSB_HOME", request.Config.RuntimeHome) + request.Config.RuntimeHome = home guard, err := allocationLock(request) if err != nil { t.Fatal(err) @@ -335,49 +342,58 @@ func TestPermanentRestoreImportFailureKeepsOneAdmission(t *testing.T) { } } -func TestUndispatchedFailureClosesOnlyExplicitCancellation(t *testing.T) { - for _, cause := range []error{context.Canceled, fmt.Errorf("wrapped: %w", context.DeadlineExceeded), sandbox.ErrOwnership, sandbox.ErrInvalid, os.ErrNotExist, errors.New("opaque SDK failure")} { - t.Run(cause.Error(), func(t *testing.T) { - request := initialInfoRequest(t) - guard, err := allocationLock(request) - if err != nil { - t.Fatal(err) - } - defer guard.Close() - b := backend{q: request} - q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "exact-target", Generation: 1}} - if _, err = b.admitRestore(q); err != nil { - t.Fatal(err) - } - if err = b.closeUndispatchedRestore(q, cause); err != nil { - t.Fatal(err) - } - guard.Close() - guard, err = allocationLock(request) - if err != nil { - t.Fatal(err) - } - defer guard.Close() - q.ObserveOnly = true - closed, err := b.admitRestore(q) - want := errors.Is(cause, context.Canceled) || errors.Is(cause, context.DeadlineExceeded) - if err != nil || closed != want { - t.Fatal("wrong closure", closed, want, err) - } - q.ObserveOnly = false - closed, err = b.admitRestore(q) - if want { - if err != nil || !closed { - t.Fatal("late dispatch permitted", closed, err) - } - } else if !errors.Is(err, wire.ErrUnconfirmed) { - t.Fatal("old admission replayed", err) - } - }) +func TestFreshRestoreRequestDeadlineClosesBeforeNativeExecution(t *testing.T) { + if !isolatedSDKTest(t) { + return } + request := initialInfoRequest(t) + t.Setenv("MSB_HOME", request.Config.RuntimeHome) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + // The real request acquired its lock before its deadline. Time can expire + // while prior SDK work settles; the lifecycle context deliberately stays live. + request.Deadline = time.Now().Add(-time.Second) + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: wire.Name(request.Config, request.Reference, 1), Generation: 1}} + request.Operation = "resume" + request.Resume = &q + b := backend{q: request} + _, err = b.run(context.Background()) + if !errors.Is(err, context.DeadlineExceeded) { + t.Fatal("request deadline not enforced", err) + } + q.ObserveOnly = true + state, err := b.resume(context.Background(), q) + if err != nil || state.RestoreAttemptClosed != q.OperationID || state.Status != "absent" { + t.Fatal("closed proof missing", state, err) + } + q.ObserveOnly = false + if _, err = b.resume(context.Background(), q); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("late request replayed", err) + } +} + +// The official SDK freezes its default backend on first use. Match production's +// one-request helper process rather than changing MSB_HOME under a cached pool. +func isolatedSDKTest(t *testing.T) bool { + t.Helper() + if os.Getenv("OAC_NATIVE_TEST_CHILD") == t.Name() { + return true + } + cmd := exec.Command(os.Args[0], "-test.run=^"+t.Name()+"$", "-test.count=1") + cmd.Env = append(os.Environ(), "OAC_NATIVE_TEST_CHILD="+t.Name()) + if output, err := cmd.CombinedOutput(); err != nil { + t.Fatalf("isolated SDK test: %v\n%s", err, output) + } + return false } func TestAdmittedAbsentObservationSkipsArchiveAndStaysUnknown(t *testing.T) { + if !isolatedSDKTest(t) { + return + } request := initialInfoRequest(t) t.Setenv("MSB_HOME", request.Config.RuntimeHome) guard, err := allocationLock(request) @@ -397,6 +413,7 @@ func TestAdmittedAbsentObservationSkipsArchiveAndStaysUnknown(t *testing.T) { if err != nil { t.Fatal(err) } + b.q.Deadline = time.Now().Add(-time.Second) q.ObserveOnly = true state, err := b.resume(t.Context(), q) if !errors.Is(err, wire.ErrUnconfirmed) || state.RestoreAttemptClosed != "" { @@ -421,6 +438,7 @@ func TestPreDispatchClosureRejectsConflictingJournal(t *testing.T) { t.Fatal(err) } defer guard.Close() + request.Deadline = time.Now().Add(-time.Second) b := backend{q: request} q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "target", Generation: 1}} r := restoreAdmission{OperationID: q.OperationID, Target: q.Target, Snapshot: q.Snapshot} @@ -440,9 +458,56 @@ func TestPreDispatchClosureRejectsConflictingJournal(t *testing.T) { t.Fatal(err) } } - if err = b.closeUndispatchedRestore(q, context.Canceled); err == nil { - t.Fatal("unsafe closure accepted") + before, readErr := os.ReadFile(path) + wantErr := sandbox.ErrOwnership + if kind == "missing" { + wantErr = os.ErrNotExist + if !errors.Is(readErr, os.ErrNotExist) { + t.Fatal("expected missing journal", readErr) + } + } else if readErr != nil { + t.Fatal(readErr) + } + if err = b.closeExpiredRestore(q); !errors.Is(err, wantErr) { + t.Fatalf("unsafe closure result: got %v, want %v", err, wantErr) + } + after, readErr := os.ReadFile(path) + if kind == "missing" { + if !errors.Is(readErr, os.ErrNotExist) { + t.Fatal("missing journal was created", readErr) + } + } else if readErr != nil || string(after) != string(before) { + t.Fatal("rejected closure changed journal", readErr) } }) } } + +func TestFreshAdmissionClosesOnlyAfterRequestDeadline(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "target", Generation: 1}} + if _, err = b.admitRestore(q); err != nil { + t.Fatal(err) + } + if err = b.closeExpiredRestore(q); err != nil { + t.Fatal("live request was closed", err) + } + var r restoreAdmission + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + if err = readPrivateJSON(path, &r); err != nil || r.Closed { + t.Fatal(r, err) + } + b.q.Deadline = time.Now().Add(-time.Second) + if err = b.closeExpiredRestore(q); !errors.Is(err, context.DeadlineExceeded) { + t.Fatal(err) + } + if err = readPrivateJSON(path, &r); err != nil || !r.Closed { + t.Fatal("expired dispatch not fenced", r, err) + } +} From 473dd6f7b5c0be22aa09131388e8add7809ce7de Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Sat, 10 Oct 2026 19:15:23 +0000 Subject: [PATCH 5/6] Accept checkpoint qualification in node health JSON --- .../internal/sandbox/node/generation_json.go | 10 +++- .../sandbox/node/generation_json_test.go | 57 +++++++++++++++++++ 2 files changed, 65 insertions(+), 2 deletions(-) diff --git a/services/core/internal/sandbox/node/generation_json.go b/services/core/internal/sandbox/node/generation_json.go index 5336a320f..06ae10fff 100644 --- a/services/core/internal/sandbox/node/generation_json.go +++ b/services/core/internal/sandbox/node/generation_json.go @@ -62,9 +62,15 @@ func generationRecords(raw []byte, required, optional string) error { return sandbox.ErrInvalid } for _, entry := range entries { - if _, err := generationObject(entry, required, optional, ""); err != nil { + fields, err := generationObject(entry, required, optional, "") + if err != nil { return err } + if checkpoint := fields["checkpoint"]; checkpoint != nil { + if _, err := generationObject(checkpoint, "artifact_domain execution_class", "", ""); err != nil { + return err + } + } } return nil } @@ -110,7 +116,7 @@ func validateGenerationJSON(raw []byte, kind string) error { if err != nil { return err } - if err := generationRecords(health["generations"], "generation specification_digest state", "diagnostic"); err != nil { + if err := generationRecords(health["generations"], "generation specification_digest state", "diagnostic checkpoint"); err != nil { return err } } diff --git a/services/core/internal/sandbox/node/generation_json_test.go b/services/core/internal/sandbox/node/generation_json_test.go index 506b1066b..2f0f7e069 100644 --- a/services/core/internal/sandbox/node/generation_json_test.go +++ b/services/core/internal/sandbox/node/generation_json_test.go @@ -2,6 +2,7 @@ package node import ( "encoding/json" + "fmt" "strings" "testing" "time" @@ -91,3 +92,59 @@ func TestNodeGenerationManagementIsAnExplicitCurrentCapability(t *testing.T) { } } } + +func TestCheckpointGenerationHealthJSONRoundTrip(t *testing.T) { + for _, managed := range []bool{false, true} { + for _, kind := range []string{"hello", "heartbeat"} { + t.Run(fmt.Sprintf("managed=%t/%s", managed, kind), func(t *testing.T) { + compatibility := sandbox.CheckpointCompatibility{ArtifactDomain: "private-store", ExecutionClass: "native-class"} + f := frame{Version: ProtocolVersion, Type: kind, Health: &Health{ProviderReady: true, ObservedAt: time.Now().UTC(), Generations: []sandbox.GenerationStatus{{Generation: 5, SpecificationDigest: strings.Repeat("a", 64), State: "ready", Checkpoint: &compatibility}}}} + if kind == "hello" { + f.Identity = new(Identity) + f.GenerationManagement = managed + } else { + f.ConnectionID = uuid.NewString() + f.OwnerEpoch = 1 + } + raw, err := json.Marshal(f) + if err != nil { + t.Fatal(err) + } + if err := validateVersionFrame(f, len(raw)); err != nil { + t.Fatal("writer rejected valid status", err) + } + decoded, err := decodeFrame(raw) + if err != nil { + t.Fatal("decoder rejected valid status", err) + } + if decoded.Health == nil || len(decoded.Health.Generations) != 1 || decoded.Health.Generations[0].Checkpoint == nil || *decoded.Health.Generations[0].Checkpoint != compatibility { + t.Fatal("checkpoint qualification lost", decoded) + } + original := string(raw) + for name, replacement := range map[string]string{ + "unknown nested field": strings.Replace(original, `"artifact_domain":"private-store"`, `"artifact_domain":"private-store","extra":"value"`, 1), + "unknown generation field": strings.Replace(original, `"checkpoint":`, `"extra":true,"checkpoint":`, 1), + "duplicate token": strings.Replace(original, `"artifact_domain":"private-store"`, `"artifact_domain":"private-store","artifact_domain":"other"`, 1), + "case alias": strings.Replace(original, `"artifact_domain":`, `"Artifact_Domain":`, 1), + "missing token": strings.Replace(original, `"artifact_domain":"private-store",`, "", 1), + "null token": strings.Replace(original, `"artifact_domain":"private-store"`, `"artifact_domain":null`, 1), + "null checkpoint": strings.Replace(original, `{"artifact_domain":"private-store","execution_class":"native-class"}`, `null`, 1), + "empty domain": strings.Replace(original, `"private-store"`, `""`, 1), + "empty execution": strings.Replace(original, `"native-class"`, `""`, 1), + "oversized domain": strings.Replace(original, `"private-store"`, `"`+strings.Repeat("x", 257)+`"`, 1), + "oversized execution": strings.Replace(original, `"native-class"`, `"`+strings.Repeat("x", 257)+`"`, 1), + "nonready checkpoint": strings.Replace(original, `"state":"ready"`, `"state":"preparing"`, 1), + } { + t.Run(name, func(t *testing.T) { + if replacement == original { + t.Fatal("fixture unchanged") + } + if _, err := decodeFrame([]byte(replacement)); err == nil { + t.Fatal("invalid checkpoint accepted") + } + }) + } + }) + } + } +} From 48cb2aff9b2e7f08cbdbc73301ae4f3ea026df6a Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Sat, 10 Oct 2026 19:55:15 +0000 Subject: [PATCH 6/6] Document checkpoint compatibility during node maintenance --- docs/getting-started/nodes.md | 2 ++ docs/zh/getting-started/nodes.md | 4 +++- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/docs/getting-started/nodes.md b/docs/getting-started/nodes.md index e35850ca5..aa02e573c 100644 --- a/docs/getting-started/nodes.md +++ b/docs/getting-started/nodes.md @@ -14,6 +14,8 @@ You add a node by generating a command in Web and running it on the host. The [s For microsandbox checkpoint recovery across nodes, add `--checkpoint-root /absolute/private/shared-store` to the installer command on each participating node. Mount the same private store at those paths before installation and grant only the node service account access. The default is a private node-local store and supports recovery on that node. The checkpoint store is separate from workspace storage and must never be guest-accessible; the [checkpoint transfer contract](../sandbox-provider.md#checkpoint-transfer) owns compatibility and retention requirements. +Microsandbox includes the host kernel and CPU profile in its checkpoint execution compatibility. An OS or CPU-profile change can therefore make retained checkpoints incompatible even on their original node. Before maintenance, keep a node with the matching execution class and the same ready generation available for those checkpoints; otherwise their requests wait under the [checkpoint retention policy](../sandbox-provider.md#checkpoint-transfer). + For microsandbox with independent workspace storage, complete the [NFS mount and service-account setup](../configuration.md#independent-workspace-storage) on this host before enrollment. The node receives the selected immutable filesystem configuration with each binding; do not author a separate node storage setting. The Core host joins like any other host: to run sandboxes on it, add it as a node. diff --git a/docs/zh/getting-started/nodes.md b/docs/zh/getting-started/nodes.md index bf4135da6..9e7d24e05 100644 --- a/docs/zh/getting-started/nodes.md +++ b/docs/zh/getting-started/nodes.md @@ -1,7 +1,7 @@ --- title: "添加和管理节点" source: docs/getting-started/nodes.md -source_hash: a31edf7908ecd0a848b562bbceda3fd872436473d5ec1f2bedfba8719ac9e2a1 +source_hash: 0c62518cc740387ed03035ea4870108edeaaa82331a7509b573fb2b775ae3aba --- 节点是一台 Linux 主机,在沙箱后端为 Docker 或 microsandbox 时,为 Core 托管 Session 运行沙箱。Core 将新 Session 分配给有空余容量的节点;节点创建沙箱,沙箱回连 Core。E2B 不需要节点。应用为自己的 Session 连接的机器是[自托管执行器](self-hosted.md),而不是节点。 @@ -18,6 +18,8 @@ Core 主机与其他主机一样加入:要在它上面运行沙箱,将它添 microsandbox 跨节点检查点恢复要求在每个参与节点的安装命令加入 `--checkpoint-root /absolute/private/shared-store`。安装前将同一个私有 store 挂载到这些路径,并只允许 node 服务账户访问。默认使用节点本地私有 store,只支持在该节点恢复。检查点 store 独立于 workspace 存储,绝不能允许 guest 访问;[检查点转移契约](../sandbox-provider.md#checkpoint-transfer) 定义兼容性和保留要求。 +Microsandbox 的检查点执行兼容性包含主机内核和 CPU profile。因此,操作系统或 CPU profile 变化可能使保留的检查点不再兼容,甚至无法在原节点恢复。维护前,应为这些检查点保留具有匹配 execution class 且已就绪同一 generation 的节点;否则,相关请求会按[检查点保留策略](../sandbox-provider.md#checkpoint-transfer)等待。 + 对于使用独立工作区存储的 microsandbox,应在注册前完成此主机上的 [NFS 挂载和服务账户配置](../configuration.md#independent-workspace-storage)。节点随每次 binding 接收所选不可变文件系统配置;不要单独编写节点存储设置。 ## 添加节点 {#add-a-node}