diff --git a/.github/workflows/e2b-template.yml b/.github/workflows/e2b-template.yml index 9a0100767..461e874c1 100644 --- a/.github/workflows/e2b-template.yml +++ b/.github/workflows/e2b-template.yml @@ -14,6 +14,11 @@ on: type: string required: false + resume_build: + description: Existing unsubmitted templateID:build_UUID after an upload failure; requires ref + type: string + required: false + permissions: contents: read @@ -63,6 +68,7 @@ jobs: env: SOURCE_BRANCH: ${{ inputs.source_branch }} REQUESTED_REF: ${{ inputs.ref }} + RESUME_BUILD: ${{ inputs.resume_build }} run: | set -euo pipefail case "$SOURCE_BRANCH" in main|beta) ;; *) exit 1 ;; esac @@ -70,6 +76,10 @@ jobs: echo '::error::ref must be a full lowercase commit SHA.' exit 1 fi + if [ -n "$RESUME_BUILD" ]; then + [ -n "$REQUESTED_REF" ] || { echo '::error::Recovery requires the original source SHA.'; exit 1; } + [[ "$RESUME_BUILD" =~ ^[a-zA-Z0-9_-]+:[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$ ]] || exit 1 + fi git fetch --no-tags origin "refs/heads/$SOURCE_BRANCH:refs/remotes/origin/$SOURCE_BRANCH" revision="$(git rev-parse "${REQUESTED_REF:-origin/$SOURCE_BRANCH}^{commit}")" git merge-base --is-ancestor "$revision" "origin/$SOURCE_BRANCH" || { @@ -146,6 +156,8 @@ jobs: printf '%s\n' "$image" > "$output_root/runtime-image.id" - name: Build one immutable E2B template env: + RESUME_BUILD: ${{ inputs.resume_build }} + SOURCE_COMMIT: ${{ steps.source.outputs.revision }} E2B_API_KEY: ${{ secrets.E2B_API_KEY }} E2B_API_URL: ${{ vars.E2B_API_URL }} run: | @@ -156,10 +168,76 @@ jobs: printf '%s' "$E2B_API_KEY" > "$key" unset E2B_API_KEY mkdir -p "$RUNNER_TEMP/e2b-result" - "$RUNNER_TEMP/e2b-sdk/bin/python" source/services/core/deploy/e2b/build-template.py \ + "$RUNNER_TEMP/e2b-sdk/bin/python" - source/services/core/deploy/e2b/build-template.py \ --image "$(cat "$HOME/.oac/build/e2b-runtime/runtime-image.id")" \ --name "sandbase-oac-$GITHUB_RUN_ID-$GITHUB_RUN_ATTEMPT" \ - --api-key-file "$key" --output "$RUNNER_TEMP/e2b-result/template.json" + --api-key-file "$key" --output "$RUNNER_TEMP/e2b-result/template.json" <<'PY' + import hashlib, json, os, pathlib, runpy, sys + from types import SimpleNamespace + from e2b.template.main import TemplateBuilder + import e2b.template_sync.main as sdk + + output = pathlib.Path(os.environ['RUNNER_TEMP']) / 'e2b-result' + original_copy = TemplateBuilder.copy + def copy(self, src, dest, *args, **kwargs): + if str(src) != 'runtime.tar.gz': + return original_copy(self, src, dest, *args, **kwargs) + if dest != '/root/runtime.tar.gz' or args or kwargs != {'user': 'root'}: + raise RuntimeError('Runtime archive copy contract changed') + context = pathlib.Path(self._template._file_context_path) + digest = hashlib.sha256() + parts = [] + with (context / src).open('rb') as stream: + for index in range(256): + block = stream.read(32 * 1024 * 1024) + if not block: + break + digest.update(block) + name = f'runtime.part{index:04d}' + (context / name).write_bytes(block) + self = original_copy(self, name, '/root/' + name, user='root') + parts.append('/root/' + name) + if stream.read(1): + raise RuntimeError('Runtime archive exceeds 8 GiB') + if not parts: + raise RuntimeError('Runtime archive is empty') + (output / 'upload.json').write_text(json.dumps({ + 'chunks': len(parts), 'chunk_bytes': 32 * 1024 * 1024, + 'runtime_sha256': digest.hexdigest()}, indent=2) + '\n') + command = ('cat ' + ' '.join(parts) + ' > /root/runtime.tar.gz && ' + 'echo "' + digest.hexdigest() + ' /root/runtime.tar.gz" | sha256sum -c - && ' + 'rm ' + ' '.join(parts)) + return self.run_cmd(command, user='root') + TemplateBuilder.copy = copy + + original_request = sdk.request_build + def request(client, **kwargs): + selector = os.environ.get('RESUME_BUILD', '') + if selector: + template_id, build_id = selector.split(':') + # Inspect the raw response before the SDK's typed parser. + response = client.get_httpx_client().get( + f'/templates/{template_id}/builds/{build_id}/status') + response.raise_for_status() + state = response.json() + if (state.get('templateID') != template_id or state.get('buildID') != build_id + or state.get('status') != 'waiting' or state.get('logEntries') + or state.get('logs')): + raise RuntimeError('Recovery requires an unsubmitted waiting build with no logs') + result = SimpleNamespace(template_id=template_id, build_id=build_id, tags=[]) + else: + result = original_request(client, **kwargs) + receipt = {'template': result.template_id + ':' + result.build_id, + 'source_commit': os.environ['SOURCE_COMMIT'], + 'api_url': os.environ['E2B_API_URL'], 'resumed': bool(selector)} + (output / 'build-request.json').write_text(json.dumps(receipt, indent=2) + '\n') + return result + sdk.request_build = request + builder = sys.argv[1] + sys.path.insert(0, str(pathlib.Path(builder).resolve().parent)) + sys.argv = sys.argv[1:] + runpy.run_path(builder, run_name='__main__') + PY - name: Record the build identity env: SOURCE_COMMIT: ${{ steps.source.outputs.revision }} @@ -179,8 +257,9 @@ jobs: out.write('One combined Runtime template built. No Core configuration or production deployment was changed.\n') PY - uses: actions/upload-artifact@v6 + if: always() with: name: e2b-template-${{ steps.source.outputs.revision }}-${{ github.run_id }}-${{ github.run_attempt }} - path: ${{ runner.temp }}/e2b-result/template.json - if-no-files-found: error + path: ${{ runner.temp }}/e2b-result/*.json + if-no-files-found: warn retention-days: 90 diff --git a/Makefile b/Makefile index 5a93f513b..9de40a1a8 100644 --- a/Makefile +++ b/Makefile @@ -166,23 +166,20 @@ build-mcode-runtime: .PHONY: build-microsandbox-provider check-microsandbox-provider build-microsandbox-provider: - @test "$$(go env GOOS)" = linux || { echo 'The microsandbox provider helper requires Linux' >&2; exit 1; } - @set -e; output="$${OAC_DEV_HOME:-$$HOME/.oac}/build/microsandbox-provider"; \ - [[ "$$output" == /* ]] || { echo 'Provider output directory must be absolute' >&2; exit 1; }; \ - mkdir -p "$$output"; \ - cd services/core/tools/microsandbox-provider; \ - GOWORK=off CGO_ENABLED=1 go build -mod=readonly -trimpath -o "$$output/oac-microsandbox-provider" . + python3 scripts/build-microsandbox-provider.py build "$${OAC_DEV_HOME:-$$HOME/.oac}/build/microsandbox-provider/oac-microsandbox-provider" check-microsandbox-provider: + python3 scripts/build-microsandbox-provider.test.py go test -mod=readonly ./services/core/internal/sandbox/microsandbox/... -count=1 @if [[ "$$(go env GOOS)" == linux ]]; then \ - cd services/core/tools/microsandbox-provider && GOWORK=off CGO_ENABLED=1 go test -mod=readonly ./... -count=1; \ + python3 scripts/build-microsandbox-provider.py test; \ else \ printf 'Skipping the Linux-only microsandbox SDK helper tests; the full Linux gate is required before release.\n'; \ fi .PHONY: check-distribution build-core-distribution check-distribution: + PYTHONDONTWRITEBYTECODE=1 python3 services/core/deploy/runtime_profile_test.py node --test scripts/build-native-catalog.test.mjs go test ./services/web ./services/core/cmd/oac -count=1 PYTHONDONTWRITEBYTECODE=1 python3 -m unittest discover -s deploy/node -p 'test_*.py' diff --git a/README.ja.md b/README.ja.md new file mode 100644 index 000000000..b5c2e3cf6 --- /dev/null +++ b/README.ja.md @@ -0,0 +1,78 @@ +
+ +![OpenAgentCore — One core. Many agents.](docs/assets/openagentcore-banner.jpeg) + +# OpenAgentCore + +複数のネイティブ実行エンジンに対応し、自分のインフラにデプロイできる OpenAI Agents API のオープンソース実装です。 + +[公式サイト](https://openagentcore.dev/) · [インストール](#インストール) · [API を使う](https://openagentcore.dev/docs/getting-started/quickstart) · [ドキュメント](https://openagentcore.dev/docs/getting-started/) · [コントリビューション](CONTRIBUTING.md) + +[English](README.md) · [简体中文](README.zh-CN.md) · **日本語** + +
+ +## コンポーネントの関係 + +![OpenAgentCore のアーキテクチャ](docs/assets/architecture.png) + +アプリケーションと管理者は、次の Core API を使用します。 + +| API | パス | 利用者 | +| --- | --- | --- | +| **[Agents API](https://openagentcore.dev/docs/api/public-agent-api)** | `/v1` | アプリケーション。[OpenAI の Agents API](https://developers.openai.com/api/docs/guides/agents-api/overview) と同じプロトコルを使用 | +| **[Core API](https://openagentcore.dev/contracts/agents-api/admin-api)** | `/core/v1` | 管理者。Web 経由で利用 | + +Core は実行状態を永続化し、Runtime は Environment 内で選択した Harness を実行します。各コンポーネントは定義されたプロトコルで接続されているため、個別に置き換えられます。詳しくは[アーキテクチャガイド](https://openagentcore.dev/docs/architecture)をご覧ください。 + +## OpenAgentCore とは + +OpenAgentCore は、自分のインフラ上で AI エージェントを実行し、[OpenAI Agents API](https://developers.openai.com/api/docs/guides/agents-api/overview) を提供します。 + +- **OpenAI と同じ API。** [公式 OpenAI SDK](https://developers.openai.com/api/docs/guides/agents/sdk) または直接の HTTP リクエストで、接続先を変更するだけで利用できます。新しいクライアントを覚える必要はありません。 +- **エージェントを選択可能。** 各 [Session](https://openagentcore.dev/docs/api/public-agent-api) は、[Codex](https://github.com/openai/codex)、[Claude Code](https://code.claude.com/docs/en/overview)、[MiniMax Code](https://github.com/MiniMax-AI/minimax-code) のいずれかの[ネイティブ Harness](https://openagentcore.dev/contracts/agents-api/harness-onboarding) を実行し、[設定したモデルプロバイダー](https://openagentcore.dev/contracts/agents-api/model-execution)を使用します。 +- **実行するマシンを選択可能。** エージェントは、[マネージドサンドボックス](https://openagentcore.dev/contracts/agents-api/sandbox-deployment)([Docker](https://www.docker.com/)、[microsandbox](https://github.com/zerocore-ai/microsandbox)、[E2B](https://e2b.dev/))でも、自分の Linux、macOS、Windows マシンでも動作します。 +- **すべてのコンポーネントを置き換え可能。** [サンドボックス](https://openagentcore.dev/docs/sandbox-provider)、[Harness](https://openagentcore.dev/contracts/agents-api/harness-onboarding)、[モデルプロバイダー](https://openagentcore.dev/contracts/agents-api/model-execution)は、[定義されたプロトコル](AGENTS.md#protocols-at-every-boundary)を通じて接続されます。 + +## スクリーンショット + +| 概要 | エージェントの監視 | +| --- | --- | +| ![デプロイの概要](docs/assets/console-overview-en.webp) | ![エージェントの監視](docs/assets/console-agent-metrics-en.webp) | + +## インストール + +[前提条件](https://openagentcore.dev/docs/getting-started/install#prerequisites)に従って Docker をセットアップした Linux または macOS で、次のコマンドを実行します。 + +```sh +curl -fsSL https://github.com/MiniMax-AI/OpenAgentCore/releases/latest/download/install.sh | bash +``` + +Windows PowerShell の場合: + +```powershell +irm https://github.com/MiniMax-AI/OpenAgentCore/releases/latest/download/install.ps1 | iex +``` + +続いて、以下の手順で設定します。 + +1. インストーラーが生成した Core key で **Web(管理コンソール)にサインイン**し、**ドメインと HTTPS を設定**します。 +2. **デフォルトモデルを設定**し、**Project API key を発行**します。 +3. ノード、E2B、または自分のマシンを**実行リソースとして追加**します。 +4. OpenAI SDK で **[最初の Session を実行](https://openagentcore.dev/docs/getting-started/quickstart)** します。 + +[インストールガイド](https://openagentcore.dev/docs/getting-started/install)では、各手順に加え、HTTPS の設定とローカルで手軽に試す方法を説明しています。リッスンアドレスやポートなどの設定は、[インストールオプション](https://openagentcore.dev/docs/getting-started/install-options)をご覧ください。 + +## ドキュメント + +| 目的 | 参照先 | +| --- | --- | +| インストールと運用 | [インストールガイド](https://openagentcore.dev/docs/getting-started/install)、続いて[運用ガイド](https://openagentcore.dev/docs/getting-started/operations) | +| API を使ったアプリケーション開発 | [クイックスタート](https://openagentcore.dev/docs/getting-started/quickstart)、続いて [Agents API ガイド](https://openagentcore.dev/docs/api/public-agent-api) | +| 完成したアプリケーションの確認 | [サンプル](https://openagentcore.dev/docs/examples) | +| 自分のマシンでエージェントを実行 | [セルフホスト実行](https://openagentcore.dev/docs/getting-started/self-hosted) | +| Harness の機能と制限の確認 | [Harness の機能](https://openagentcore.dev/contracts/agents-api/harness-capabilities) | +| 設計の理解 | [アーキテクチャガイド](https://openagentcore.dev/docs/architecture) | +| 新しいサンドボックス、Harness などのコンポーネントの追加 | [開発ガイド](https://openagentcore.dev/docs/development) | + +すべてのドキュメントは[ドキュメント一覧](https://openagentcore.dev/docs/getting-started/)から参照できます。コードを変更する前に、[コントリビューションガイド](CONTRIBUTING.md)をお読みください。 diff --git a/README.md b/README.md index d8056f965..bf0a5187f 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ An open-source, self-hosted implementation of the OpenAI Agents API with multipl [Website](https://openagentcore.dev/) · [Install](#install) · [Call the API](https://openagentcore.dev/docs/getting-started/quickstart) · [Documentation](https://openagentcore.dev/docs/getting-started/) · [Contributing](CONTRIBUTING.md) -**English** · [简体中文](README.zh-CN.md) +**English** · [简体中文](README.zh-CN.md) · [日本語](README.ja.md) diff --git a/README.zh-CN.md b/README.zh-CN.md index fb55a62a5..8d632d49a 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -8,7 +8,7 @@ OpenAI Agents API 的开源实现,支持多种原生执行引擎,可部署 [官网](https://openagentcore.dev/zh/) · [安装](#安装) · [调用 API](https://openagentcore.dev/zh/docs/getting-started/quickstart) · [文档](https://openagentcore.dev/zh/docs/getting-started/) · [参与贡献](CONTRIBUTING.md) -[English](README.md) · **简体中文** +[English](README.md) · **简体中文** · [日本語](README.ja.md) diff --git a/apps/daemon/internal/agent/claudesdk/declaration.go b/apps/daemon/internal/agent/claudesdk/declaration.go index d422248fa..c1d903d98 100644 --- a/apps/daemon/internal/agent/claudesdk/declaration.go +++ b/apps/daemon/internal/agent/claudesdk/declaration.go @@ -24,6 +24,7 @@ var Declaration = agent.Declaration{Info: proto.SupportedAgentKind{Kind: "claude Streaming: proto.CapabilitySupported, Usage: proto.CapabilitySupported, Resume: proto.CapabilitySupported, + RetainedNativeHistory: proto.CapabilityUnsupported, NativeSessionRecovery: proto.CapabilityUnsupported, Steering: proto.CapabilitySupported, MessageItems: proto.CapabilitySupported, @@ -68,18 +69,17 @@ func discoverWithCheck(parent context.Context, options agent.DiscoveryOptions, d if !filepath.IsAbs(entrypoint) { return fail(fmt.Errorf("%s must be absolute", claudeSDKEntrypointEnv)) } - profileDir, err := paths.ProfileDir(options.Profile) - if err != nil { + if err := paths.ValidateProfile(options.Profile); err != nil { return fail(err) } - if !filepath.IsAbs(profileDir) { - return fail(fmt.Errorf("Claude SDK state requires an absolute OAC_RUNTIME_HOME")) + if !filepath.IsAbs(options.RuntimeRoot) || filepath.Clean(options.RuntimeRoot) != options.RuntimeRoot { + return fail(fmt.Errorf("Claude SDK requires an explicit Runtime root")) } node := os.Getenv(claudeSDKNodeEnv) if node == "" { node = "node" } - node, err = exec.LookPath(node) + node, err := exec.LookPath(node) if err != nil { return fail(fmt.Errorf("Claude SDK Node executable is unavailable")) } @@ -87,21 +87,20 @@ func discoverWithCheck(parent context.Context, options agent.DiscoveryOptions, d if err != nil { return fail(err) } - config = Config{Node: node, Entrypoint: entrypoint, StateDir: filepath.Join(profileDir, "runtime", "claude-sdk")} + if !filepath.IsAbs(options.StateRoot) || filepath.Clean(options.StateRoot) != options.StateRoot { + return fail(fmt.Errorf("Claude SDK requires an explicit state root")) + } + config = Config{Node: node, Entrypoint: entrypoint, StateDir: filepath.Join(options.StateRoot, "daemon", options.Profile, "runtime", "claude-sdk")} binding, err := localworkspace.Load() if err != nil { return fail(err) } if binding != nil { - root, err := paths.Root() - if err != nil { - return fail(err) - } config.Node, err = filepath.EvalSymlinks(node) if err != nil { return fail(err) } - config, err = ConfigureLocal(config, root, os.Getenv("OAC_RUNTIME_WORKSPACE"), binding.NetworkPolicy()) + config, err = ConfigureLocal(config, options.StateRoot, options.RuntimeRoot, os.Getenv("OAC_RUNTIME_WORKSPACE"), binding.NetworkPolicy()) if err != nil { return fail(err) } @@ -118,6 +117,7 @@ func discoverWithCheck(parent context.Context, options agent.DiscoveryOptions, d caps.EnvironmentNone, caps.FunctionTools = proto.CapabilityUnsupported, proto.CapabilityFromBool(info.SupportsWorkspaceFunctions()) caps.LocalEnvironment, caps.WorkspaceReadPreparation = proto.CapabilitySupported, proto.CapabilitySupported caps.NativeSessionRecovery = proto.CapabilitySupported + caps.RetainedNativeHistory = proto.CapabilityFromBool(info.SDK == "0.3.269" && info.Native == "2.1.269 (Claude Code)") } out.Info.Available, out.Info.Version = true, info.SDK out.Info.Capabilities.MessageImages = proto.CapabilityFromBool(info.SupportsMessageImages()) diff --git a/apps/daemon/internal/agent/claudesdk/declaration_test.go b/apps/daemon/internal/agent/claudesdk/declaration_test.go index 9181b5454..de5271b7b 100644 --- a/apps/daemon/internal/agent/claudesdk/declaration_test.go +++ b/apps/daemon/internal/agent/claudesdk/declaration_test.go @@ -24,9 +24,9 @@ func TestClaudeSDKInvalidPathsFailBeforeProbe(t *testing.T) { if relative == "entrypoint" { t.Setenv(claudeSDKEntrypointEnv, "main.js") } else { - t.Setenv("OAC_RUNTIME_HOME", "relative-home") + root = "relative-home" } - out := discoverWithCheck(t.Context(), agent.DiscoveryOptions{Profile: "default", Stdout: &strings.Builder{}, Stderr: &strings.Builder{}}, Declaration.Info, func(context.Context, Config) (RuntimeInfo, error) { + out := discoverWithCheck(t.Context(), agent.DiscoveryOptions{RuntimeRoot: root, StateRoot: t.TempDir(), Profile: "default", Stdout: &strings.Builder{}, Stderr: &strings.Builder{}}, Declaration.Info, func(context.Context, Config) (RuntimeInfo, error) { t.Fatal("invalid paths reached runtime probe") return RuntimeInfo{}, nil }) @@ -47,7 +47,7 @@ func TestClaudeSDKFeatureDiscovery(t *testing.T) { } t.Setenv(claudeSDKNodeEnv, node) for _, features := range [][]string{nil, {"mcp_http_tools"}, {"mcp_http_bearer_auth"}, {"mcp_http_tools", "mcp_http_bearer_auth"}, {"mcp_http_required"}, {"mcp_http_tools", "mcp_http_required"}, {"subagent_resources"}, {"structured_output"}} { - out := discoverWithCheck(t.Context(), agent.DiscoveryOptions{Profile: "default", Stdout: &strings.Builder{}, Stderr: &strings.Builder{}}, Declaration.Info, func(context.Context, Config) (RuntimeInfo, error) { + out := discoverWithCheck(t.Context(), agent.DiscoveryOptions{RuntimeRoot: root, StateRoot: t.TempDir(), Profile: "default", Stdout: &strings.Builder{}, Stderr: &strings.Builder{}}, Declaration.Info, func(context.Context, Config) (RuntimeInfo, error) { info := RuntimeInfo{SDK: "0.3.269", Native: "2.1.269 (Claude Code)", Features: features} return info, nil }) @@ -92,7 +92,7 @@ func TestRuntimeDiscoveryConfigurationAndRegistration(t *testing.T) { } t.Setenv(claudeSDKNodeEnv, node) entrypoint := filepath.Join(root, "bundle", "main.js") - options := agent.DiscoveryOptions{Profile: "test", Stdout: io.Discard, Stderr: io.Discard} + options := agent.DiscoveryOptions{RuntimeRoot: root, StateRoot: t.TempDir(), Profile: "test", Stdout: io.Discard, Stderr: io.Discard} for _, configured := range []bool{false, true} { for _, ready := range []bool{false, true} { t.Setenv(claudeSDKEntrypointEnv, "") @@ -102,7 +102,7 @@ func TestRuntimeDiscoveryConfigurationAndRegistration(t *testing.T) { calls := 0 runtime := discoverWithCheck(t.Context(), options, Declaration.Info, func(_ context.Context, c Config) (RuntimeInfo, error) { calls++ - if c.Node != node || c.Entrypoint != entrypoint || c.StateDir != filepath.Join(root, "daemon", "test", "runtime", "claude-sdk") || c.Env != nil { + if c.Node != node || c.Entrypoint != entrypoint || c.StateDir != filepath.Join(options.StateRoot, "daemon", "test", "runtime", "claude-sdk") || c.Env != nil { t.Fatalf("configuration: %+v", c) } if !ready { diff --git a/apps/daemon/internal/agent/claudesdk/local.go b/apps/daemon/internal/agent/claudesdk/local.go index d3340f3f9..d8bf94c47 100644 --- a/apps/daemon/internal/agent/claudesdk/local.go +++ b/apps/daemon/internal/agent/claudesdk/local.go @@ -11,8 +11,11 @@ import ( // ConfigureLocal selects the qualified, dedicated Runtime layout. The shared // localworkspace binding still authorizes every request against its Session. -func ConfigureLocal(config Config, root, workspace string, network agentnetwork.Policy) (Config, error) { - config.StateDir = filepath.Join(root, "runtime", "claude-sdk", "history") +func ConfigureLocal(config Config, stateRoot, root, workspace string, network agentnetwork.Policy) (Config, error) { + if !filepath.IsAbs(stateRoot) || filepath.Clean(stateRoot) != stateRoot { + return Config{}, fmt.Errorf("claudesdk: state root must be a clean absolute directory") + } + config.StateDir = filepath.Join(stateRoot, "runtime", "claude-sdk", "history") config.Workspace = &WorkspaceConfig{ Directory: workspace, PublicDirectory: workspace, NetworkAccess: network.Access, AllowedDomains: network.Hosts(), HomeDir: filepath.Join(root, "runtime", "claude-sdk", "home"), diff --git a/apps/daemon/internal/agent/claudesdk/local_test.go b/apps/daemon/internal/agent/claudesdk/local_test.go index 2d6408645..797562ebc 100644 --- a/apps/daemon/internal/agent/claudesdk/local_test.go +++ b/apps/daemon/internal/agent/claudesdk/local_test.go @@ -9,6 +9,7 @@ import ( "testing" "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" + "github.com/MiniMax-AI/OpenAgentCore/internal/agentnetwork" ) func TestLocalWorkspaceBindingNetworkAndRequiredHistory(t *testing.T) { @@ -82,3 +83,25 @@ func TestWorkspaceProviderCredentialsReplaceAmbientSelection(t *testing.T) { } } } + +func TestLocalHistorySurvivesChangingInstanceRoot(t *testing.T) { + config := workspaceFixture(t) + retained, instance := t.TempDir(), t.TempDir() + first, err := ConfigureLocal(config, retained, instance, config.Workspace.Directory, agentnetwork.Policy{Access: "enabled"}) + if err != nil { + t.Fatal(err) + } + second, err := ConfigureLocal(config, retained, t.TempDir(), config.Workspace.Directory, agentnetwork.Policy{Access: "enabled"}) + if err != nil { + t.Fatal(err) + } + if first.StateDir != second.StateDir || first.StateDir != filepath.Join(retained, "runtime", "claude-sdk", "history") { + t.Fatal("history moved with compute") + } + if first.Workspace.HomeDir == second.Workspace.HomeDir || first.Workspace.ScratchDir == second.Workspace.ScratchDir { + t.Fatal("instance HOME or scratch retained across compute") + } + if _, err := ConfigureLocal(config, "relative", instance, config.Workspace.Directory, agentnetwork.Policy{Access: "enabled"}); err == nil { + t.Fatal("invalid state root accepted") + } +} diff --git a/apps/daemon/internal/agent/claudesdk/options.go b/apps/daemon/internal/agent/claudesdk/options.go index 46ffde096..81f968eff 100644 --- a/apps/daemon/internal/agent/claudesdk/options.go +++ b/apps/daemon/internal/agent/claudesdk/options.go @@ -6,7 +6,6 @@ import ( "path/filepath" "strings" - "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/paths" "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" harnessconfiguration "github.com/MiniMax-AI/OpenAgentCore/internal/harnessconfig/claudesdk" ) @@ -152,13 +151,8 @@ func prepareConfiguration(config Config, req proto.PromptRequestPayload) (startR if req.LocalEnvironment != nil || req.RequireExistingNativeSession { return fail("local execution and history recovery require a dedicated workspace") } - root, err := paths.Root() - if err != nil { - return startRequest{}, nil, err - } - relative, err := filepath.Rel(root, config.StateDir) - if err != nil || !filepath.IsAbs(root) || !filepath.IsAbs(config.StateDir) || relative == "." || relative == ".." || strings.HasPrefix(relative, ".."+string(filepath.Separator)) { - return fail("SDK state must be in a managed runtime subdirectory") + if !filepath.IsAbs(config.StateDir) || filepath.Clean(config.StateDir) != config.StateDir || filepath.Dir(config.StateDir) == config.StateDir { + return fail("SDK state must be an explicit private absolute directory") } start.Cwd = filepath.Join(config.StateDir, "work") for _, dir := range []string{config.StateDir, filepath.Join(config.StateDir, "tmp"), start.Cwd} { diff --git a/apps/daemon/internal/agent/claudesdk/session_test.go b/apps/daemon/internal/agent/claudesdk/session_test.go index 30899cfc7..7c0f094c2 100644 --- a/apps/daemon/internal/agent/claudesdk/session_test.go +++ b/apps/daemon/internal/agent/claudesdk/session_test.go @@ -86,7 +86,7 @@ func TestTextFactoryRejectsUnsupportedInput(t *testing.T) { case "tool": request.FunctionTools = []proto.FunctionTool{{}} case "outside": - config.StateDir = filepath.Dir(root) + config.StateDir = "relative-state" } _, err := startSingleTurn(context.Background(), config, request, make(chan proto.Envelope, 1)) if err == nil || !strings.HasPrefix(err.Error(), "claudesdk:") { diff --git a/apps/daemon/internal/agent/codex/declaration.go b/apps/daemon/internal/agent/codex/declaration.go index 14d3277a0..6d02affce 100644 --- a/apps/daemon/internal/agent/codex/declaration.go +++ b/apps/daemon/internal/agent/codex/declaration.go @@ -18,6 +18,7 @@ var Declaration = agent.Declaration{Info: proto.SupportedAgentKind{ Streaming: proto.CapabilitySupported, Usage: proto.CapabilitySupported, Resume: proto.CapabilitySupported, + RetainedNativeHistory: proto.CapabilityUnsupported, NativeSessionRecovery: proto.CapabilityUnsupported, Steering: proto.CapabilitySupported, MessageItems: proto.CapabilitySupported, @@ -62,8 +63,9 @@ func discoverWithCheck(parent context.Context, options agent.DiscoveryOptions, i caps.NativeSessionRecovery = proto.CapabilityFromBool(SupportsNativeSessionRecovery(version)) caps.LocalEnvironment = proto.CapabilityFromBool(SupportsLocalEnvironment(version)) caps.WorkspaceReadPreparation = caps.LocalEnvironment + caps.RetainedNativeHistory = caps.LocalEnvironment caps.MCPHTTPRequired = proto.CapabilityFromBool(SupportsNativeSessionRecovery(version)) - runtime.Executor = NewExecutorFactory() + runtime.Executor = NewExecutorFactory(options.StateRoot) fmt.Fprintf(options.Stdout, "Codex preflight ok (%s)\n", version) return runtime } diff --git a/apps/daemon/internal/agent/codex/error_classification_test.go b/apps/daemon/internal/agent/codex/error_classification_test.go index cc6e21340..6a5a04014 100644 --- a/apps/daemon/internal/agent/codex/error_classification_test.go +++ b/apps/daemon/internal/agent/codex/error_classification_test.go @@ -50,7 +50,7 @@ func TestClassificationOnlyFromRootTerminalError(t *testing.T) { for _, mode := range []string{"failed", "recovered", "notification-only"} { t.Run(mode, func(t *testing.T) { out := make(chan proto.Envelope, 16) - s := &Session{runID: "run", out: out, cancelCtx: context.Background(), cfg: defaultSessionConfig(), rpc: NewJSONRPCClient(JSONRPCConfig{})} + s := &Session{runID: "run", out: out, cancelCtx: context.Background(), cfg: defaultSessionConfig(testStateRoot(t)), rpc: NewJSONRPCClient(JSONRPCConfig{})} s.registerHandlers() s.setThreadID("root") scopeNotification(t, s, "turn/started", `{"threadId":"root","turn":{"id":"turn"}}`) diff --git a/apps/daemon/internal/agent/codex/execution_controls_test.go b/apps/daemon/internal/agent/codex/execution_controls_test.go index e9d6f5451..374e0e1a0 100644 --- a/apps/daemon/internal/agent/codex/execution_controls_test.go +++ b/apps/daemon/internal/agent/codex/execution_controls_test.go @@ -11,7 +11,7 @@ func TestExecutionControlsSelectNativeSettings(t *testing.T) { t.Setenv("OAC_RUNTIME_HOME", t.TempDir()) for _, search := range []string{"disabled", "cached", "live"} { for _, verbosity := range []string{"low", "medium", "high"} { - plan, err := BuildSessionPlan("state", map[string]any{"model": "test-model"}, &proto.ExecutionControls{WebSearch: search, TextVerbosity: verbosity}) + plan, err := BuildSessionPlan(testStateRoot(t), "state", map[string]any{"model": "test-model"}, &proto.ExecutionControls{WebSearch: search, TextVerbosity: verbosity}) if err != nil { t.Fatal(err) } @@ -30,7 +30,7 @@ func TestExecutionControlsRejectIncompleteOrInvalidValues(t *testing.T) { {}, {WebSearch: "disabled"}, {TextVerbosity: "medium"}, {WebSearch: "invalid", TextVerbosity: "medium"}, {WebSearch: "disabled", TextVerbosity: "invalid"}, } { - if plan, err := BuildSessionPlan("state", nil, &controls); err == nil { + if plan, err := BuildSessionPlan(testStateRoot(t), "state", nil, &controls); err == nil { plan.Cleanup() t.Fatal("invalid controls accepted", controls) } diff --git a/apps/daemon/internal/agent/codex/executor.go b/apps/daemon/internal/agent/codex/executor.go index d8997e525..81a545a7c 100644 --- a/apps/daemon/internal/agent/codex/executor.go +++ b/apps/daemon/internal/agent/codex/executor.go @@ -21,8 +21,8 @@ type Executor struct { closeMu sync.Mutex } -func PrepareExecutor(ctx context.Context, req proto.PromptRequestPayload) (agent.Executor, error) { - return newExecutor(ctx, req, defaultSessionConfig()) +func PrepareExecutor(ctx context.Context, stateRoot string, req proto.PromptRequestPayload) (agent.Executor, error) { + return newExecutor(ctx, req, defaultSessionConfig(stateRoot)) } func newExecutor(ctx context.Context, req proto.PromptRequestPayload, cfg sessionConfig) (*Executor, error) { @@ -172,4 +172,8 @@ func (e *Executor) Close(ctx context.Context) error { var _ agent.Executor = (*Executor)(nil) -func NewExecutorFactory() agent.ExecutorFactory { return PrepareExecutor } +func NewExecutorFactory(stateRoot string) agent.ExecutorFactory { + return func(ctx context.Context, req proto.PromptRequestPayload) (agent.Executor, error) { + return PrepareExecutor(ctx, stateRoot, req) + } +} diff --git a/apps/daemon/internal/agent/codex/executor_native_test.go b/apps/daemon/internal/agent/codex/executor_native_test.go index f14f1340e..faee82904 100644 --- a/apps/daemon/internal/agent/codex/executor_native_test.go +++ b/apps/daemon/internal/agent/codex/executor_native_test.go @@ -45,7 +45,7 @@ func TestExecutorNativeReuse(t *testing.T) { for _, name := range []string{"CODEX_EXEC_SERVER_URL", "CODEX_EXEC_SERVER_NOISE_REGISTRY_URL", "CODEX_EXEC_SERVER_NOISE_ENVIRONMENT_ID", "CODEX_EXEC_SERVER_NOISE_AUTH_TOKEN"} { t.Setenv(name, "") } - cfg := defaultSessionConfig() + cfg := defaultSessionConfig(testStateRoot(t)) cfg.codexBinary = binary cfg.logger = slog.New(slog.DiscardHandler) req := proto.PromptRequestPayload{ diff --git a/apps/daemon/internal/agent/codex/harness_config_test.go b/apps/daemon/internal/agent/codex/harness_config_test.go index cd3577d3c..7ca086aa9 100644 --- a/apps/daemon/internal/agent/codex/harness_config_test.go +++ b/apps/daemon/internal/agent/codex/harness_config_test.go @@ -13,7 +13,7 @@ import ( func TestHarnessConfigAppliedWithoutChangingProvider(t *testing.T) { t.Setenv("OAC_RUNTIME_HOME", t.TempDir()) - plan, err := BuildSessionPlan("native-config", map[string]any{ + plan, err := BuildSessionPlan(testStateRoot(t), "native-config", map[string]any{ "model": "fixture", "harness_config": map[string]any{"model_reasoning_effort": "high"}, "model_provider": map[string]any{"base_url": "https://provider.invalid/v1", "protocol": "responses", "api_key": "test-key"}, }, nil) @@ -27,7 +27,7 @@ func TestHarnessConfigAppliedWithoutChangingProvider(t *testing.T) { } func TestHarnessConfigConflictFailsBeforePreparation(t *testing.T) { - _, err := BuildSessionPlan("", map[string]any{"harness_config": map[string]any{"model_provider": "bypass"}}, nil) + _, err := BuildSessionPlan(testStateRoot(t), "", map[string]any{"harness_config": map[string]any{"model_provider": "bypass"}}, nil) if err != harnessconfig.ErrHarnessConfig { t.Fatalf("configuration must fail before filesystem preparation: %v", err) } diff --git a/apps/daemon/internal/agent/codex/mcp_http_bearer_test.go b/apps/daemon/internal/agent/codex/mcp_http_bearer_test.go index 0223e6716..5a68ce270 100644 --- a/apps/daemon/internal/agent/codex/mcp_http_bearer_test.go +++ b/apps/daemon/internal/agent/codex/mcp_http_bearer_test.go @@ -22,7 +22,7 @@ func TestMCPHTTPBearerPlanSeparatesServersAndProcesses(t *testing.T) { req := proto.PromptRequestPayload{AgentStateKey: "retained-mcp", DisableExecutionEnvironment: true, MCPHTTPServers: &servers} seen := map[string]bool{} for range 2 { - plan, _, err := prepareSessionPlan(t.Context(), req, defaultSessionConfig()) + plan, _, err := prepareSessionPlan(t.Context(), req, defaultSessionConfig(testStateRoot(t))) if err != nil { t.Fatal(err) } @@ -58,7 +58,7 @@ func TestMCPHTTPBearerRejectsInvalidTokensWithoutPersistence(t *testing.T) { for _, token := range []string{"", "=", " has-space", "has-space ", "has space", "line\r\ninjection", "nul\x00byte", "opaque中文", "middle=padding", "punctuation:invalid"} { servers := []proto.MCPHTTPServer{{ConnectionOrigin: "service", ServerLabel: "tools", ServerURL: "https://tools.example/mcp", BearerToken: &token}} req := proto.PromptRequestPayload{AgentStateKey: "invalid-bearer", DisableExecutionEnvironment: true, MCPHTTPServers: &servers} - if _, _, err := prepareSessionPlan(t.Context(), req, defaultSessionConfig()); err == nil || err.Error() != "invalid HTTPS MCP bearer credential" { + if _, _, err := prepareSessionPlan(t.Context(), req, defaultSessionConfig(testStateRoot(t))); err == nil || err.Error() != "invalid HTTPS MCP bearer credential" { t.Fatal("invalid bearer value accepted or unsafe error returned") } } @@ -82,7 +82,7 @@ func TestMCPHTTPBearerDoesNotReachModelCatalogProbe(t *testing.T) { servers := []proto.MCPHTTPServer{{ConnectionOrigin: "service", ServerLabel: "tools", ServerURL: "https://tools.example/mcp", BearerToken: &token}} req := proto.PromptRequestPayload{AgentStateKey: "catalog", DisableExecutionEnvironment: true, MCPHTTPServers: &servers, AgentOptions: map[string]any{"model": "fixture-model"}, ExecutionControls: &proto.ExecutionControls{WebSearch: "disabled", TextVerbosity: "medium"}} - cfg := defaultSessionConfig() + cfg := defaultSessionConfig(testStateRoot(t)) cfg.codexBinary = binary plan, _, err := prepareSessionPlan(t.Context(), req, cfg) if err != nil { diff --git a/apps/daemon/internal/agent/codex/mcp_http_preflight_test.go b/apps/daemon/internal/agent/codex/mcp_http_preflight_test.go index c57e3297c..138a97182 100644 --- a/apps/daemon/internal/agent/codex/mcp_http_preflight_test.go +++ b/apps/daemon/internal/agent/codex/mcp_http_preflight_test.go @@ -166,7 +166,7 @@ func TestPublicMCPHTTPPreparationChecksBeforeNewAndResumedThread(t *testing.T) { t.Fatal(err) } assertPreparationOnly(t, root) - home, err := allocCodexHome(req.AgentStateKey) + home, err := allocCodexHome(testStateRoot(t), req.AgentStateKey) if err != nil { t.Fatal(err) } diff --git a/apps/daemon/internal/agent/codex/mcp_http_test.go b/apps/daemon/internal/agent/codex/mcp_http_test.go index 20ac1767f..015c0e939 100644 --- a/apps/daemon/internal/agent/codex/mcp_http_test.go +++ b/apps/daemon/internal/agent/codex/mcp_http_test.go @@ -21,7 +21,7 @@ func TestPublicMCPHTTPPlanOwnsConfigurationAndPreservesHistory(t *testing.T) { {ConnectionOrigin: "service", ServerLabel: "blocked", ServerURL: "http://127.0.0.1:12345/mcp", AllowedTools: &denyAll}, } req := proto.PromptRequestPayload{AgentStateKey: "public-mcp", DisableExecutionEnvironment: true, MCPHTTPServers: &servers} - plan, _, err := prepareSessionPlan(t.Context(), req, defaultSessionConfig()) + plan, _, err := prepareSessionPlan(t.Context(), req, defaultSessionConfig(testStateRoot(t))) if err != nil { t.Fatal(err) } @@ -29,7 +29,7 @@ func TestPublicMCPHTTPPlanOwnsConfigurationAndPreservesHistory(t *testing.T) { if !slices.Contains(plan.DisableFeatures, "apps") || !slices.Contains(plan.DisableFeatures, "plugins") || !slices.Contains(plan.ExtraConfig, [2]string{"mcp_oauth_credentials_store", `"file"`}) { t.Fatal("native profile was not pinned") } - home, err := allocCodexHome(req.AgentStateKey) + home, err := allocCodexHome(testStateRoot(t), req.AgentStateKey) if err != nil { t.Fatal(err) } @@ -54,7 +54,7 @@ func TestPublicMCPHTTPPlanOwnsConfigurationAndPreservesHistory(t *testing.T) { t.Fatal(err) } servers = []proto.MCPHTTPServer{{ConnectionOrigin: "service", ServerLabel: "replacement", ServerURL: "https://new.example/mcp"}} - second, _, err := prepareSessionPlan(t.Context(), req, defaultSessionConfig()) + second, _, err := prepareSessionPlan(t.Context(), req, defaultSessionConfig(testStateRoot(t))) if err != nil { t.Fatal(err) } @@ -90,7 +90,7 @@ func TestPublicMCPHTTPRejectsInvalidProfileAndStoredCredentials(t *testing.T) { } } t.Setenv("OAC_RUNTIME_HOME", t.TempDir()) - home, err := allocCodexHome("credentials") + home, err := allocCodexHome(testStateRoot(t), "credentials") if err != nil { t.Fatal(err) } @@ -100,7 +100,7 @@ func TestPublicMCPHTTPRejectsInvalidProfileAndStoredCredentials(t *testing.T) { t.Fatal(err) } req := proto.PromptRequestPayload{AgentStateKey: "credentials", DisableExecutionEnvironment: true, MCPHTTPServers: &valid} - if _, _, err := prepareSessionPlan(t.Context(), req, defaultSessionConfig()); err == nil { + if _, _, err := prepareSessionPlan(t.Context(), req, defaultSessionConfig(testStateRoot(t))); err == nil { t.Fatal("existing MCP credentials accepted") } if after, err := os.ReadFile(path); err != nil || !reflect.DeepEqual(after, stored) { diff --git a/apps/daemon/internal/agent/codex/model_route_test.go b/apps/daemon/internal/agent/codex/model_route_test.go index 32aeb4680..c18fecfe9 100644 --- a/apps/daemon/internal/agent/codex/model_route_test.go +++ b/apps/daemon/internal/agent/codex/model_route_test.go @@ -11,7 +11,7 @@ func TestPlanRejectsNonNativeFrozenProvider(t *testing.T) { for _, protocol := range []string{"anthropic", "chat_completions"} { t.Run(protocol, func(t *testing.T) { provider := map[string]any{"protocol": protocol, "base_url": "https://model.invalid/v1", "api_key": "private-sentinel"} - plan, err := BuildSessionPlan("frozen-state", map[string]any{"model": "frozen-model", "model_provider": provider}, nil) + plan, err := BuildSessionPlan(testStateRoot(t), "frozen-state", map[string]any{"model": "frozen-model", "model_provider": provider}, nil) if plan.Cleanup != nil { plan.Cleanup() } @@ -30,7 +30,7 @@ func TestPlanRejectsIncompleteExplicitProvider(t *testing.T) { {"model": "chosen", "model_provider": nil}, {"model_provider": map[string]any{"protocol": "responses", "base_url": "https://model.example/v1", "api_key": "fixture"}}, } { - if _, err := BuildSessionPlan("state", options, nil); err == nil { + if _, err := BuildSessionPlan(testStateRoot(t), "state", options, nil); err == nil { t.Fatal("explicit provider fell back to native defaults") } } diff --git a/apps/daemon/internal/agent/codex/model_verbosity_test.go b/apps/daemon/internal/agent/codex/model_verbosity_test.go index e1b9ecd45..83a65e502 100644 --- a/apps/daemon/internal/agent/codex/model_verbosity_test.go +++ b/apps/daemon/internal/agent/codex/model_verbosity_test.go @@ -37,7 +37,7 @@ func TestPrepareModelVerbosity(t *testing.T) { if err := os.WriteFile(binary, []byte("#!/bin/sh\nprintf '%s' '"+catalog+"'\n"), 0700); err != nil { t.Fatal(err) } - plan, err := BuildSessionPlan("state", map[string]any{"model": "known-model"}, &proto.ExecutionControls{WebSearch: "disabled", TextVerbosity: "high"}) + plan, err := BuildSessionPlan(testStateRoot(t), "state", map[string]any{"model": "known-model"}, &proto.ExecutionControls{WebSearch: "disabled", TextVerbosity: "high"}) if err != nil { t.Fatal(err) } @@ -80,7 +80,7 @@ func TestPrepareDefaultModelVerbosity(t *testing.T) { for _, model := range []string{"supported", "unsupported", "unknown-provider-model"} { for _, level := range []string{"low", "medium", "high"} { t.Run(model+"/"+level, func(t *testing.T) { - plan, err := BuildSessionPlan("state", map[string]any{"model": model}, &proto.ExecutionControls{WebSearch: "disabled", TextVerbosity: level}) + plan, err := BuildSessionPlan(testStateRoot(t), "state", map[string]any{"model": model}, &proto.ExecutionControls{WebSearch: "disabled", TextVerbosity: level}) if err != nil { t.Fatal(err) } diff --git a/apps/daemon/internal/agent/codex/options.go b/apps/daemon/internal/agent/codex/options.go index 1f53a8d24..02c1e194e 100644 --- a/apps/daemon/internal/agent/codex/options.go +++ b/apps/daemon/internal/agent/codex/options.go @@ -9,7 +9,6 @@ import ( "slices" "strings" - "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/paths" "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" harnessconfiguration "github.com/MiniMax-AI/OpenAgentCore/internal/harnessconfig/codex" ) @@ -71,7 +70,7 @@ type SessionPlan struct { // // Execution controls select the native web_search and model_verbosity settings. // The codex binary itself is resolved via PATH only. -func BuildSessionPlan(agentStateKey string, opts map[string]any, controls *proto.ExecutionControls) (SessionPlan, error) { +func BuildSessionPlan(stateRoot, agentStateKey string, opts map[string]any, controls *proto.ExecutionControls) (SessionPlan, error) { plan := SessionPlan{ // Harnesses run unattended: Codex never offers its ask-the-user tool. ExtraConfig: [][2]string{{"tools.experimental_request_user_input.enabled", "false"}}, @@ -99,7 +98,7 @@ func BuildSessionPlan(agentStateKey string, opts map[string]any, controls *proto plan.Model = stringOpt(opts, "model") plan.SystemPrompt = stringOpt(opts, "system_prompt") - codexHome, err := allocCodexHome(agentStateKey) + codexHome, err := allocCodexHome(stateRoot, agentStateKey) if err != nil { return plan, err } @@ -117,6 +116,11 @@ func BuildSessionPlan(agentStateKey string, opts map[string]any, controls *proto return plan, err } plan.ModelProvider = oacProviderSlug + // The pinned native shell snapshot captures the loop environment before + // shell_environment_policy is applied. Do not persist provider credentials. + plan.DisableFeatures = append(plan.DisableFeatures, "shell_snapshot") + env = append(env, providerAPIKeyEnv+"="+provider.BearerToken) + plan.ExtraConfig = append(plan.ExtraConfig, [2]string{"shell_environment_policy.exclude", "[" + strconv(providerAPIKeyEnv) + "]"}) } plan.Env = env @@ -142,18 +146,17 @@ func BuildSessionPlan(agentStateKey string, opts map[string]any, controls *proto // helpers // --------------------------------------------------------------------------- -func allocCodexHome(agentStateKey string) (string, error) { +func allocCodexHome(root, agentStateKey string) (string, error) { if strings.TrimSpace(agentStateKey) == "" { return "", fmt.Errorf("codex: agentStateKey required for CODEX_HOME allocation") } - root, err := paths.Root() - if err != nil { - return "", err + if !filepath.IsAbs(root) || filepath.Clean(root) != root { + return "", fmt.Errorf("codex: state root must be a clean absolute directory") } parts := strings.Split(agentStateKey, "/") safeParts := make([]string, 0, len(parts)) for _, part := range parts { - if safe := safePathPartCodex(part); safe != "" { + if safe := safePathPartCodex(part); safe != "" && safe != "." && safe != ".." { safeParts = append(safeParts, safe) } } diff --git a/apps/daemon/internal/agent/codex/options_test.go b/apps/daemon/internal/agent/codex/options_test.go index cb8910aaa..eb64d84de 100644 --- a/apps/daemon/internal/agent/codex/options_test.go +++ b/apps/daemon/internal/agent/codex/options_test.go @@ -8,7 +8,7 @@ import ( ) func TestBuildSessionPlan_DefaultsToBypass(t *testing.T) { - plan, err := BuildSessionPlan("conv-1/agent-1/codex", nil, nil) + plan, err := BuildSessionPlan(testStateRoot(t), "conv-1/agent-1/codex", nil, nil) if err != nil { t.Fatalf("BuildSessionPlan: %v", err) } @@ -22,7 +22,7 @@ func TestBuildSessionPlan_DefaultsToBypass(t *testing.T) { } func TestBuildSessionPlan_AllocsCodexHomeAndEnv(t *testing.T) { - plan, err := BuildSessionPlan("conv-1/agent-1/codex", nil, nil) + plan, err := BuildSessionPlan(testStateRoot(t), "conv-1/agent-1/codex", nil, nil) if err != nil { t.Fatalf("BuildSessionPlan: %v", err) } @@ -47,11 +47,11 @@ func TestBuildSessionPlan_AllocsCodexHomeAndEnv(t *testing.T) { func TestBuildSessionPlan_StableCodexHomeByStateKey(t *testing.T) { stateKey := "conv-stable/agent-stable/codex" - planA, err := BuildSessionPlan(stateKey, nil, nil) + planA, err := BuildSessionPlan(testStateRoot(t), stateKey, nil, nil) if err != nil { t.Fatalf("BuildSessionPlan A: %v", err) } - planB, err := BuildSessionPlan(stateKey, nil, nil) + planB, err := BuildSessionPlan(testStateRoot(t), stateKey, nil, nil) if err != nil { t.Fatalf("BuildSessionPlan B: %v", err) } diff --git a/apps/daemon/internal/agent/codex/permission_profile_test.go b/apps/daemon/internal/agent/codex/permission_profile_test.go index e69720d6e..c8bd52e71 100644 --- a/apps/daemon/internal/agent/codex/permission_profile_test.go +++ b/apps/daemon/internal/agent/codex/permission_profile_test.go @@ -12,7 +12,7 @@ func TestRuntimeUsesHostPermissions(t *testing.T) { t.Setenv("OAC_RUNTIME_HOME", t.TempDir()) { req := proto.PromptRequestPayload{AgentStateKey: "session", DisableSubagents: true, LocalEnvironment: &proto.LocalEnvironment{NetworkAccess: "enabled", WorkspaceRoot: t.TempDir()}} - plan, _, err := prepareSessionPlan(t.Context(), req, sessionConfig{}) + plan, _, err := prepareSessionPlan(t.Context(), req, sessionConfig{stateRoot: testStateRoot(t)}) if err != nil { t.Fatal(err) } @@ -27,7 +27,7 @@ func TestRuntimeUsesHostPermissions(t *testing.T) { plan.Cleanup() for _, network := range []string{"disabled", "restricted"} { req.LocalEnvironment.NetworkAccess = network - if _, _, err := prepareSessionPlan(t.Context(), req, sessionConfig{}); err == nil { + if _, _, err := prepareSessionPlan(t.Context(), req, sessionConfig{stateRoot: testStateRoot(t)}); err == nil { t.Fatal("unsupported network admitted") } } @@ -49,7 +49,7 @@ func TestSelfHostedToolEnvironmentCannotRedirectNativeHistory(t *testing.T) { t.Setenv(key, value) } req := proto.PromptRequestPayload{AgentStateKey: "session", DisableSubagents: true, LocalEnvironment: &proto.LocalEnvironment{NetworkAccess: "enabled"}} - plan, _, err := prepareSessionPlan(t.Context(), req, sessionConfig{}) + plan, _, err := prepareSessionPlan(t.Context(), req, sessionConfig{stateRoot: testStateRoot(t)}) if err != nil { t.Fatal(err) } diff --git a/apps/daemon/internal/agent/codex/preparation_helpers_test.go b/apps/daemon/internal/agent/codex/preparation_helpers_test.go index 2749dc255..1472c3a5d 100644 --- a/apps/daemon/internal/agent/codex/preparation_helpers_test.go +++ b/apps/daemon/internal/agent/codex/preparation_helpers_test.go @@ -37,7 +37,7 @@ func preparationFixture(t *testing.T) (proto.PromptRequestPayload, sessionConfig if err := os.WriteFile(binary, []byte(body), 0o700); err != nil { t.Fatal(err) } - cfg := defaultSessionConfig() + cfg := defaultSessionConfig(testStateRoot(t)) cfg.codexBinary = binary req := proto.PromptRequestPayload{ AgentKind: "codex", AgentStateKey: "prepared-session", diff --git a/apps/daemon/internal/agent/codex/provider_config.go b/apps/daemon/internal/agent/codex/provider_config.go index b3c571e0d..9eca33486 100644 --- a/apps/daemon/internal/agent/codex/provider_config.go +++ b/apps/daemon/internal/agent/codex/provider_config.go @@ -16,6 +16,7 @@ import ( // the config.toml deterministic and removes a footgun where two prompts // in the same CODEX_HOME could disagree on which provider to use. const oacProviderSlug = "oac" +const providerAPIKeyEnv = "OAC_CODEX_PROVIDER_API_KEY" // providerConfig is the daemon-internal view of the model provider the // Runtime prepared for this Session. Flattened from the common @@ -30,8 +31,7 @@ type providerConfig struct { // BaseURL is the model provider's HTTPS endpoint, including any // path prefix (e.g. /v1). Required. BaseURL string - // BearerToken is the literal API key. Codex's `experimental_bearer_token` - // field accepts a string; we emit it verbatim. Required. + // BearerToken is supplied only through the native env_key mechanism. BearerToken string // HTTPHeaders is forwarded as `[model_providers.oac.http_headers]`. // Keys / values rendered as TOML basic strings. @@ -90,8 +90,8 @@ func writeCodexProviderConfig(codexHome string, cfg providerConfig) error { b.WriteString(tomlQuoteString(cfg.BaseURL)) b.WriteByte('\n') - b.WriteString("experimental_bearer_token = ") - b.WriteString(tomlQuoteString(cfg.BearerToken)) + b.WriteString("env_key = ") + b.WriteString(tomlQuoteString(providerAPIKeyEnv)) b.WriteByte('\n') wireAPI := strings.TrimSpace(cfg.WireAPI) @@ -159,7 +159,7 @@ func writeCodexProviderConfig(codexHome string, cfg providerConfig) error { // // File is opened O_APPEND so concurrent writers in the same prompt // (today: at most one of each) don't race. 0o600 perms because the -// file carries the API bearer token in plaintext. +// file contains private per-Session configuration. func appendConfigTOML(path string, body string) error { f, err := os.OpenFile(path, os.O_CREATE|os.O_APPEND|os.O_WRONLY, 0o600) if err != nil { diff --git a/apps/daemon/internal/agent/codex/provider_config_test.go b/apps/daemon/internal/agent/codex/provider_config_test.go index 80eb8d48b..9ca848612 100644 --- a/apps/daemon/internal/agent/codex/provider_config_test.go +++ b/apps/daemon/internal/agent/codex/provider_config_test.go @@ -3,6 +3,7 @@ package codex import ( "os" "path/filepath" + "slices" "strings" "testing" ) @@ -21,7 +22,7 @@ func TestWriteCodexProviderConfig_Minimal(t *testing.T) { `[model_providers.oac]`, `name = "OpenAgentCore"`, `base_url = "https://platform-api.example.com/v1"`, - `experimental_bearer_token = "sk-test"`, + `env_key = "OAC_CODEX_PROVIDER_API_KEY"`, `wire_api = "responses"`, } { if !strings.Contains(body, want) { @@ -74,7 +75,7 @@ func TestWriteCodexProviderConfig_FullProvider(t *testing.T) { for _, want := range []string{ `name = "mygw"`, `base_url = "https://platform-api.example.com/v1"`, - `experimental_bearer_token = "sk-test-fixture"`, + `env_key = "OAC_CODEX_PROVIDER_API_KEY"`, `request_max_retries = 4`, `stream_max_retries = 3`, `[model_providers.oac.http_headers]`, @@ -156,7 +157,7 @@ func TestNormaliseProviderConfig_Nil(t *testing.T) { } func TestBuildSessionPlan_PinsModelProviderWhenProviderSet(t *testing.T) { - plan, err := BuildSessionPlan("conv-1/agent-1/codex", map[string]any{ + plan, err := BuildSessionPlan(testStateRoot(t), "conv-1/agent-1/codex", map[string]any{ "model": "fixture-model", "model_provider": map[string]any{"protocol": "responses", "base_url": "https://x/v1", @@ -194,7 +195,7 @@ func TestBuildSessionPlan_PinsModelProviderWhenProviderSet(t *testing.T) { } func TestBuildSessionPlan_NoProviderLeavesBuiltinDefault(t *testing.T) { - plan, err := BuildSessionPlan("conv-1/agent-1/codex", nil, nil) + plan, err := BuildSessionPlan(testStateRoot(t), "conv-1/agent-1/codex", nil, nil) if err != nil { t.Fatalf("BuildSessionPlan: %v", err) } @@ -217,3 +218,65 @@ func mustReadFile(t *testing.T, path string) string { } return string(b) } + +func TestProviderCredentialOnlyInNativeLoopEnvironment(t *testing.T) { + root := t.TempDir() + const token = "synthetic-private-provider-token" + plan, err := BuildSessionPlan(root, "stable-session", map[string]any{"model": "fixture-model", "model_provider": map[string]any{"protocol": "responses", "base_url": "https://model.example/v1", "api_key": token}}, nil) + if err != nil { + t.Fatal(err) + } + defer plan.Cleanup() + body := mustReadFile(t, filepath.Join(nativeHomeFromPlan(plan), "config.toml")) + if strings.Contains(body, token) || strings.Contains(body, "experimental_bearer_token") { + t.Fatal("provider credential persisted") + } + foundEnv, foundExclude := false, false + for _, entry := range plan.Env { + if entry == providerAPIKeyEnv+"="+token { + foundEnv = true + } + } + for _, entry := range plan.ExtraConfig { + if entry[0] == "shell_environment_policy.exclude" && entry[1] == `["OAC_CODEX_PROVIDER_API_KEY"]` { + foundExclude = true + } + if strings.Contains(entry[1], token) { + t.Fatal("credential exposed in native arguments") + } + } + if !foundEnv || !foundExclude || !slices.Contains(plan.DisableFeatures, "shell_snapshot") { + t.Fatal("native auth or explicit tool-environment exclusion absent") + } + if _, exists := os.LookupEnv(providerAPIKeyEnv); exists { + t.Fatal("provider credential leaked to Runtime process environment") + } +} + +func TestExplicitStateRootSurvivesRuntimeReplacement(t *testing.T) { + state := t.TempDir() + t.Setenv("OAC_RUNTIME_HOME", t.TempDir()) + first, err := allocCodexHome(state, "stable-session") + if err != nil { + t.Fatal(err) + } + history := filepath.Join(first, "history-sentinel") + if err = os.WriteFile(history, []byte("retained"), 0600); err != nil { + t.Fatal(err) + } + t.Setenv("OAC_RUNTIME_HOME", t.TempDir()) + second, err := allocCodexHome(state, "stable-session") + if err != nil || second != first { + t.Fatal("instance identity changed native history", err) + } + if body, err := os.ReadFile(history); err != nil || string(body) != "retained" { + t.Fatal("history lost", err) + } + unavailable := filepath.Join(t.TempDir(), "file") + if err = os.WriteFile(unavailable, nil, 0600); err != nil { + t.Fatal(err) + } + if _, err = allocCodexHome(unavailable, "stable-session"); err == nil { + t.Fatal("unavailable explicit root fell back") + } +} diff --git a/apps/daemon/internal/agent/codex/recovery_test.go b/apps/daemon/internal/agent/codex/recovery_test.go index ebabf06b4..c4ea41f01 100644 --- a/apps/daemon/internal/agent/codex/recovery_test.go +++ b/apps/daemon/internal/agent/codex/recovery_test.go @@ -148,7 +148,7 @@ func TestPreparedRecoveryCannotStartWithoutExistingHistory(t *testing.T) { t.Run(environment, func(t *testing.T) { req, cfg, root := preparationFixture(t) req.RequireExistingNativeSession = true - cwd, err := allocCodexHome(req.AgentStateKey) + cwd, err := allocCodexHome(testStateRoot(t), req.AgentStateKey) if err != nil { t.Fatal(err) } diff --git a/apps/daemon/internal/agent/codex/rpc_close_test.go b/apps/daemon/internal/agent/codex/rpc_close_test.go index 7b387e23a..7d4c2c29b 100644 --- a/apps/daemon/internal/agent/codex/rpc_close_test.go +++ b/apps/daemon/internal/agent/codex/rpc_close_test.go @@ -80,7 +80,7 @@ func TestJSONRPCClientCloseCanRetryUnreapedChild(t *testing.T) { closeConcurrently(true) ownerCtx, cancelOwner := context.WithCancel(t.Context()) s := &Session{rpc: client, cancelCtx: ownerCtx, cancelFn: cancelOwner, - cfg: defaultSessionConfig(), bufs: NewItemBuffers()} + cfg: defaultSessionConfig(testStateRoot(t)), bufs: NewItemBuffers()} ctx, cancelWait := context.WithTimeout(t.Context(), 20*time.Millisecond) defer cancelWait() if err := s.Cancel(ctx); !errors.Is(err, context.DeadlineExceeded) { diff --git a/apps/daemon/internal/agent/codex/session.go b/apps/daemon/internal/agent/codex/session.go index 9d3386609..044c66ae0 100644 --- a/apps/daemon/internal/agent/codex/session.go +++ b/apps/daemon/internal/agent/codex/session.go @@ -23,13 +23,15 @@ const terminalSendTimeout = 2 * time.Second // sessionConfig is the cross-cutting knob bag; production callers use // defaultSessionConfig. type sessionConfig struct { + stateRoot string codexBinary string logger *slog.Logger killTimeout time.Duration } -func defaultSessionConfig() sessionConfig { +func defaultSessionConfig(stateRoot string) sessionConfig { return sessionConfig{ + stateRoot: stateRoot, codexBinary: defaultBinary(), logger: obslog.Bg(), killTimeout: rpcKillTimeout, diff --git a/apps/daemon/internal/agent/codex/session_cancel_test.go b/apps/daemon/internal/agent/codex/session_cancel_test.go index 32e794b6a..e4307ae1e 100644 --- a/apps/daemon/internal/agent/codex/session_cancel_test.go +++ b/apps/daemon/internal/agent/codex/session_cancel_test.go @@ -116,7 +116,7 @@ func cancellationTestSession(t *testing.T) (*Session, *TestClient, ServerSide) { ctx, cancel := context.WithCancel(context.Background()) t.Cleanup(cancel) s := &Session{rpc: client.JSONRPCClient, cancelCtx: ctx, cancelFn: cancel, - cfg: defaultSessionConfig(), out: make(chan proto.Envelope, 8), bufs: NewItemBuffers()} + cfg: defaultSessionConfig(testStateRoot(t)), out: make(chan proto.Envelope, 8), bufs: NewItemBuffers()} return s, client, server } diff --git a/apps/daemon/internal/agent/codex/session_cancel_write_test.go b/apps/daemon/internal/agent/codex/session_cancel_write_test.go index 087b8e78f..e9adb4970 100644 --- a/apps/daemon/internal/agent/codex/session_cancel_write_test.go +++ b/apps/daemon/internal/agent/codex/session_cancel_write_test.go @@ -146,7 +146,7 @@ func TestCancelReleasesBlockedNativeProcess(t *testing.T) { ctx, cancel := context.WithCancel(context.Background()) defer cancel() s := &Session{rpc: client, cancelCtx: ctx, cancelFn: cancel, - cfg: defaultSessionConfig(), bufs: NewItemBuffers()} + cfg: defaultSessionConfig(testStateRoot(t)), bufs: NewItemBuffers()} s.setThreadID("native-thread") s.onTurnStarted(json.RawMessage(`{"threadId":"native-thread","turn":{"id":"native-turn"}}`)) cancelDone := make(chan struct{}) diff --git a/apps/daemon/internal/agent/codex/session_command_output_test.go b/apps/daemon/internal/agent/codex/session_command_output_test.go index a51fb6694..f14b3eab3 100644 --- a/apps/daemon/internal/agent/codex/session_command_output_test.go +++ b/apps/daemon/internal/agent/codex/session_command_output_test.go @@ -10,7 +10,7 @@ import ( func TestCommandOutputRequiresRootTurn(t *testing.T) { out := make(chan proto.Envelope, 10) - s := &Session{runID: "run", out: out, cancelCtx: context.Background(), cfg: defaultSessionConfig(), rpc: NewJSONRPCClient(JSONRPCConfig{})} + s := &Session{runID: "run", out: out, cancelCtx: context.Background(), cfg: defaultSessionConfig(testStateRoot(t)), rpc: NewJSONRPCClient(JSONRPCConfig{})} s.registerHandlers() s.setThreadID("root") s.beginRootTurn("root", "turn") diff --git a/apps/daemon/internal/agent/codex/session_messages_test.go b/apps/daemon/internal/agent/codex/session_messages_test.go index b17ed114e..1e09975fb 100644 --- a/apps/daemon/internal/agent/codex/session_messages_test.go +++ b/apps/daemon/internal/agent/codex/session_messages_test.go @@ -12,7 +12,7 @@ func TestMessageObservationIsOptInAndKeepsNativeBoundaries(t *testing.T) { for _, enabled := range []bool{false, true} { t.Run(map[bool]string{false: "legacy", true: "observed"}[enabled], func(t *testing.T) { out := make(chan proto.Envelope, 16) - s := &Session{runID: "run", observeMessages: enabled, out: out, cancelCtx: context.Background(), bufs: NewItemBuffers(), cfg: defaultSessionConfig()} + s := &Session{runID: "run", observeMessages: enabled, out: out, cancelCtx: context.Background(), bufs: NewItemBuffers(), cfg: defaultSessionConfig(testStateRoot(t))} s.setThreadID("thread") s.onTurnStarted(json.RawMessage(`{"threadId":"thread","turn":{"id":"turn"}}`)) s.onItemStarted(json.RawMessage(`{"threadId":"thread","turnId":"turn","item":{"type":"agentMessage","id":"a","phase":"commentary"}}`)) diff --git a/apps/daemon/internal/agent/codex/session_notifications_rpc_test.go b/apps/daemon/internal/agent/codex/session_notifications_rpc_test.go index f7b4d0f26..516f9bbb6 100644 --- a/apps/daemon/internal/agent/codex/session_notifications_rpc_test.go +++ b/apps/daemon/internal/agent/codex/session_notifications_rpc_test.go @@ -19,7 +19,7 @@ func TestRootRPCNotificationOrdering(t *testing.T) { ctx, cancel := context.WithTimeout(context.Background(), 3*time.Second) defer cancel() out := make(chan proto.Envelope, 16) - s := &Session{runID: "run", out: out, rpc: client.JSONRPCClient, cancelCtx: ctx, bufs: NewItemBuffers(), cfg: defaultSessionConfig()} + s := &Session{runID: "run", out: out, rpc: client.JSONRPCClient, cancelCtx: ctx, bufs: NewItemBuffers(), cfg: defaultSessionConfig(testStateRoot(t))} s.registerHandlers() barrier := make(chan struct{}, 1) client.OnNotification("test/barrier", func(json.RawMessage) { barrier <- struct{}{} }) @@ -132,7 +132,7 @@ func TestThreadRPCResponseRequiresRootIdentity(t *testing.T) { defer cleanup() ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) defer cancel() - s := &Session{rpc: client.JSONRPCClient, cancelCtx: ctx, cfg: defaultSessionConfig()} + s := &Session{rpc: client.JSONRPCClient, cancelCtx: ctx, cfg: defaultSessionConfig(testStateRoot(t))} s.registerHandlers() result := make(chan error, 1) go func() { result <- s.resumeThread("root", SessionPlan{}) }() diff --git a/apps/daemon/internal/agent/codex/session_notifications_test.go b/apps/daemon/internal/agent/codex/session_notifications_test.go index 0ee2a9b94..2c1f1092b 100644 --- a/apps/daemon/internal/agent/codex/session_notifications_test.go +++ b/apps/daemon/internal/agent/codex/session_notifications_test.go @@ -12,7 +12,7 @@ func TestRootNotificationIsolation(t *testing.T) { for _, childStarted := range []bool{false, true} { t.Run(map[bool]string{false: "child without thread started", true: "child thread started"}[childStarted], func(t *testing.T) { out := make(chan proto.Envelope, 64) - s := &Session{runID: "run", out: out, cancelCtx: context.Background(), cfg: defaultSessionConfig(), + s := &Session{runID: "run", out: out, cancelCtx: context.Background(), cfg: defaultSessionConfig(testStateRoot(t)), rpc: NewJSONRPCClient(JSONRPCConfig{}), bufs: NewItemBuffers(), observeMessages: true} s.registerHandlers() s.setThreadID("root") @@ -96,7 +96,7 @@ func TestRootNotificationIsolation(t *testing.T) { } func TestNotificationsCannotEstablishRootIdentity(t *testing.T) { - s := &Session{cfg: defaultSessionConfig(), rpc: NewJSONRPCClient(JSONRPCConfig{}), bufs: NewItemBuffers()} + s := &Session{cfg: defaultSessionConfig(testStateRoot(t)), rpc: NewJSONRPCClient(JSONRPCConfig{}), bufs: NewItemBuffers()} s.registerHandlers() scopeNotification(t, s, "thread/started", `{"thread":{"id":"unrelated"}}`) scopeNotification(t, s, "turn/started", `{"threadId":"unrelated","turn":{"id":"unrelated-turn"}}`) @@ -116,7 +116,7 @@ func TestRootNativeErrorNotification(t *testing.T) { for _, location := range []string{"error notification", "completion"} { t.Run(name+"/"+location, func(t *testing.T) { out := make(chan proto.Envelope, 4) - s := &Session{runID: "run", out: out, cancelCtx: context.Background(), cfg: defaultSessionConfig(), rpc: NewJSONRPCClient(JSONRPCConfig{})} + s := &Session{runID: "run", out: out, cancelCtx: context.Background(), cfg: defaultSessionConfig(testStateRoot(t)), rpc: NewJSONRPCClient(JSONRPCConfig{})} s.registerHandlers() s.setThreadID("root") scopeNotification(t, s, "turn/started", `{"threadId":"root","turn":{"id":"turn"}}`) diff --git a/apps/daemon/internal/agent/codex/session_plan.go b/apps/daemon/internal/agent/codex/session_plan.go index d8e7aede8..b917805f4 100644 --- a/apps/daemon/internal/agent/codex/session_plan.go +++ b/apps/daemon/internal/agent/codex/session_plan.go @@ -20,7 +20,7 @@ func prepareSessionPlan(ctx context.Context, req proto.PromptRequestPayload, cfg if err != nil { return SessionPlan{}, nil, err } - plan, err := BuildSessionPlan(req.AgentStateKey, req.AgentOptions, req.ExecutionControls) + plan, err := BuildSessionPlan(cfg.stateRoot, req.AgentStateKey, req.AgentOptions, req.ExecutionControls) if err != nil { return SessionPlan{}, nil, fmt.Errorf("codex: build session plan: %w", err) } diff --git a/apps/daemon/internal/agent/codex/session_policy_test.go b/apps/daemon/internal/agent/codex/session_policy_test.go index 86c5b3c74..f534f4cf4 100644 --- a/apps/daemon/internal/agent/codex/session_policy_test.go +++ b/apps/daemon/internal/agent/codex/session_policy_test.go @@ -16,7 +16,7 @@ func TestThreadRequestsApplyDeploymentPolicy(t *testing.T) { for _, method := range []string{"thread/start", "thread/resume"} { t.Run(method, func(t *testing.T) { t.Setenv("OAC_RUNTIME_HOME", t.TempDir()) - plan, _, err := prepareSessionPlan(context.Background(), proto.PromptRequestPayload{AgentStateKey: "conv/agent/codex", DisableSubagents: true, DisableExecutionEnvironment: true}, sessionConfig{}) + plan, _, err := prepareSessionPlan(context.Background(), proto.PromptRequestPayload{AgentStateKey: "conv/agent/codex", DisableSubagents: true, DisableExecutionEnvironment: true}, sessionConfig{stateRoot: testStateRoot(t)}) if err != nil { t.Fatal(err) } diff --git a/apps/daemon/internal/agent/codex/session_tools_test.go b/apps/daemon/internal/agent/codex/session_tools_test.go index 8d4e06c8d..460a5fe4f 100644 --- a/apps/daemon/internal/agent/codex/session_tools_test.go +++ b/apps/daemon/internal/agent/codex/session_tools_test.go @@ -29,7 +29,7 @@ func TestToolObservations(t *testing.T) { } t.Run(source.ID, func(t *testing.T) { out := make(chan proto.Envelope, 4) - s := &Session{runID: "run", out: out, cancelCtx: context.Background(), bufs: NewItemBuffers(), cfg: defaultSessionConfig()} + s := &Session{runID: "run", out: out, cancelCtx: context.Background(), bufs: NewItemBuffers(), cfg: defaultSessionConfig(testStateRoot(t))} s.setThreadID("private-thread") s.onTurnStarted(json.RawMessage(`{"threadId":"private-thread","turn":{"id":"private-turn"}}`)) raw := json.RawMessage(`{"threadId":"private-thread","turnId":"private-turn","item":` + item + `}`) diff --git a/apps/daemon/internal/agent/codex/session_usage_live_test.go b/apps/daemon/internal/agent/codex/session_usage_live_test.go index 58ccc1d44..ba9fe754f 100644 --- a/apps/daemon/internal/agent/codex/session_usage_live_test.go +++ b/apps/daemon/internal/agent/codex/session_usage_live_test.go @@ -15,7 +15,7 @@ const activeUsageNotification = `{"method":"thread/tokenUsage/updated","params": func TestUsagePublishedBeforeCompletion(t *testing.T) { out := make(chan proto.Envelope, 8) - s := &Session{runID: "run", out: out, cancelCtx: t.Context(), cfg: defaultSessionConfig()} + s := &Session{runID: "run", out: out, cancelCtx: t.Context(), cfg: defaultSessionConfig(testStateRoot(t))} s.setThreadID("thread") s.onUsageUpdated(json.RawMessage(`{"threadId":"thread","turnId":"old","tokenUsage":{"total":{"inputTokens":10,"cachedInputTokens":1,"outputTokens":3,"reasoningOutputTokens":1,"totalTokens":13}}}`)) if len(s.out) != 0 { @@ -159,7 +159,7 @@ func TestLiveUsageRequiresConsistentCompleteBreakdown(t *testing.T) { } { t.Run(name, func(t *testing.T) { out := make(chan proto.Envelope, 1) - s := &Session{out: out, cancelCtx: t.Context(), cfg: defaultSessionConfig()} + s := &Session{out: out, cancelCtx: t.Context(), cfg: defaultSessionConfig(testStateRoot(t))} s.setThreadID("thread") s.onTurnStarted(json.RawMessage(`{"threadId":"thread","turn":{"id":"turn"}}`)) s.onUsageUpdated(json.RawMessage(`{"threadId":"thread","turnId":"turn","tokenUsage":{"total":` + counters + `}}`)) @@ -177,7 +177,7 @@ func TestLiveUsageRequiresConsistentCompleteBreakdown(t *testing.T) { func TestLiveUsageRejectsInconsistentResumeDelta(t *testing.T) { out := make(chan proto.Envelope, 2) - s := &Session{out: out, cancelCtx: t.Context(), cfg: defaultSessionConfig()} + s := &Session{out: out, cancelCtx: t.Context(), cfg: defaultSessionConfig(testStateRoot(t))} s.setThreadID("thread") s.onUsageUpdated(json.RawMessage(`{"threadId":"thread","turnId":"old","tokenUsage":{"total":{"inputTokens":100,"cachedInputTokens":90,"outputTokens":50,"reasoningOutputTokens":10,"totalTokens":150}}}`)) s.onTurnStarted(json.RawMessage(`{"threadId":"thread","turn":{"id":"turn"}}`)) diff --git a/apps/daemon/internal/agent/codex/session_usage_test.go b/apps/daemon/internal/agent/codex/session_usage_test.go index bdaeec6b0..b75faf625 100644 --- a/apps/daemon/internal/agent/codex/session_usage_test.go +++ b/apps/daemon/internal/agent/codex/session_usage_test.go @@ -62,7 +62,7 @@ func TestNativeTokenUsageAcrossRuns(t *testing.T) { } func TestNativeTokenUsageFreshThreadAndIgnoredPayloads(t *testing.T) { - s := &Session{out: make(chan proto.Envelope, 8), cancelCtx: t.Context(), cfg: defaultSessionConfig()} + s := &Session{out: make(chan proto.Envelope, 8), cancelCtx: t.Context(), cfg: defaultSessionConfig(testStateRoot(t))} s.setThreadID("thread") s.onTurnStarted(json.RawMessage(`{"threadId":"thread","turn":{"id":"current"}}`)) s.onUsageUpdated(json.RawMessage(`{"threadId":"thread","turnId":"current","tokenUsage":{"total":{"inputTokens":321,"outputTokens":45}}}`)) @@ -81,7 +81,7 @@ func TestNativeTokenUsageFreshThreadAndIgnoredPayloads(t *testing.T) { } func TestLegacyTurnUsagePayload(t *testing.T) { - s := &Session{out: make(chan proto.Envelope, 8), cancelCtx: t.Context(), cfg: defaultSessionConfig()} + s := &Session{out: make(chan proto.Envelope, 8), cancelCtx: t.Context(), cfg: defaultSessionConfig(testStateRoot(t))} s.setThreadID("thread") s.onTurnStarted(json.RawMessage(`{"threadId":"thread","turn":{"id":"current"}}`)) s.onUsageUpdated(json.RawMessage(`{"threadId":"thread","usage":{"inputTokens":123,"outputTokens":12}}`)) @@ -95,7 +95,7 @@ func TestLegacyTurnUsagePayload(t *testing.T) { } func TestCompleteTokenBreakdownAndCancellation(t *testing.T) { - s := &Session{out: make(chan proto.Envelope, 8), cancelCtx: t.Context(), cfg: defaultSessionConfig()} + s := &Session{out: make(chan proto.Envelope, 8), cancelCtx: t.Context(), cfg: defaultSessionConfig(testStateRoot(t))} s.setThreadID("thread") s.onUsageUpdated(json.RawMessage(`{"threadId":"thread","turnId":"old","tokenUsage":{"total":{"inputTokens":100,"cachedInputTokens":20,"outputTokens":50,"reasoningOutputTokens":10,"totalTokens":150}}}`)) s.onTurnStarted(json.RawMessage(`{"threadId":"thread","turn":{"id":"new"}}`)) @@ -164,7 +164,7 @@ func TestUnadvancedThreadTotalIsNotTurnUsage(t *testing.T) { } { t.Run(name, func(t *testing.T) { out := make(chan proto.Envelope, 8) - s := &Session{runID: "run", out: out, cancelCtx: t.Context(), cfg: defaultSessionConfig()} + s := &Session{runID: "run", out: out, cancelCtx: t.Context(), cfg: defaultSessionConfig(testStateRoot(t))} s.setThreadID("thread") total := `{"inputTokens":0,"cachedInputTokens":0,"outputTokens":0,"reasoningOutputTokens":0,"totalTokens":0}` if replay != "" { diff --git a/apps/daemon/internal/agent/codex/state_root_test.go b/apps/daemon/internal/agent/codex/state_root_test.go new file mode 100644 index 000000000..c1b456c44 --- /dev/null +++ b/apps/daemon/internal/agent/codex/state_root_test.go @@ -0,0 +1,16 @@ +package codex + +import ( + "os" + "testing" +) + +func testStateRoot(t *testing.T) string { + t.Helper() + if root := os.Getenv("OAC_RUNTIME_HOME"); root != "" { + return root + } + root := t.TempDir() + t.Setenv("OAC_RUNTIME_HOME", root) + return root +} diff --git a/apps/daemon/internal/agent/codex/subagent_observations_test.go b/apps/daemon/internal/agent/codex/subagent_observations_test.go index 3e8f7241a..14589a919 100644 --- a/apps/daemon/internal/agent/codex/subagent_observations_test.go +++ b/apps/daemon/internal/agent/codex/subagent_observations_test.go @@ -74,7 +74,7 @@ func observationSession(t *testing.T, status string) (*Session, *subagentFixture t.Fatal(err) } f.persist(t) - s := &Session{runID: "run", nativeHome: f.home, rpc: client.JSONRPCClient, out: out, cancelCtx: ctx, cancelFn: cancel, cfg: defaultSessionConfig(), bufs: NewItemBuffers()} + s := &Session{runID: "run", nativeHome: f.home, rpc: client.JSONRPCClient, out: out, cancelCtx: ctx, cancelFn: cancel, cfg: defaultSessionConfig(testStateRoot(t)), bufs: NewItemBuffers()} s.setThreadID("root") s.beginRootTurn("root", "root-turn") s.startSubagentObservations() diff --git a/apps/daemon/internal/agent/harness.go b/apps/daemon/internal/agent/harness.go index 23f8d37e7..4804282de 100644 --- a/apps/daemon/internal/agent/harness.go +++ b/apps/daemon/internal/agent/harness.go @@ -45,7 +45,11 @@ type Declaration struct { // DiscoveryOptions provides process context without naming an implementation. type DiscoveryOptions struct { - Profile string + Profile string + // StateRoot holds private history retained with the Environment filesystem. + StateRoot string + // RuntimeRoot holds instance-private configuration and process files. + RuntimeRoot string Stdout, Stderr io.Writer } diff --git a/apps/daemon/internal/agent/mcode/declaration.go b/apps/daemon/internal/agent/mcode/declaration.go index fa22848f1..03bbef288 100644 --- a/apps/daemon/internal/agent/mcode/declaration.go +++ b/apps/daemon/internal/agent/mcode/declaration.go @@ -16,6 +16,7 @@ var Declaration = agent.Declaration{Info: proto.SupportedAgentKind{Kind: "mcode" Streaming: proto.CapabilitySupported, Usage: proto.CapabilityUnsupported, Resume: proto.CapabilitySupported, + RetainedNativeHistory: proto.CapabilityUnsupported, NativeSessionRecovery: proto.CapabilityUnsupported, Steering: proto.CapabilityUnsupported, MessageItems: proto.CapabilityUnsupported, @@ -71,7 +72,7 @@ func discoverWithCheck(parent context.Context, options agent.DiscoveryOptions, r runtime.Info = result workspace := discoverWorkspace(parent, options, runtime) if runtime.Info.Available { - runtime.Executor = NewExecutorFactory(workspace) + runtime.Executor = NewExecutorFactory(options.RuntimeRoot, workspace) } fmt.Fprintf(options.Stdout, "mcode preflight ok (%s)\n", version) return runtime diff --git a/apps/daemon/internal/agent/mcode/discovery_workspace.go b/apps/daemon/internal/agent/mcode/discovery_workspace.go index da4f768d8..aa3461192 100644 --- a/apps/daemon/internal/agent/mcode/discovery_workspace.go +++ b/apps/daemon/internal/agent/mcode/discovery_workspace.go @@ -10,7 +10,6 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/agent" "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/agent/binpath" "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/localworkspace" - "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/paths" "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" ) @@ -27,11 +26,6 @@ func discoverWorkspace(parent context.Context, options agent.DiscoveryOptions, r if binding == nil { return nil } - root, err := paths.Root() - if err != nil { - fail(err) - return nil - } node := os.Getenv("OAC_RUNTIME_MCODE_NODE") if node == "" { node = "node" @@ -56,7 +50,7 @@ func discoverWorkspace(parent context.Context, options agent.DiscoveryOptions, r fail(err) return nil } - c, err := ConfigureLocal(binary, node, os.Getenv("OAC_RUNTIME_MCODE_WORKSPACE_BRIDGE"), root, os.Getenv("OAC_RUNTIME_WORKSPACE"), binding.NetworkPolicy()) + c, err := ConfigureLocal(binary, node, os.Getenv("OAC_RUNTIME_MCODE_WORKSPACE_BRIDGE"), options.RuntimeRoot, os.Getenv("OAC_RUNTIME_WORKSPACE"), binding.NetworkPolicy()) if err == nil { err = CheckWorkspace(parent, c) } diff --git a/apps/daemon/internal/agent/mcode/environment_mcp_test.go b/apps/daemon/internal/agent/mcode/environment_mcp_test.go index 5ca8565f9..dae9ac41f 100644 --- a/apps/daemon/internal/agent/mcode/environment_mcp_test.go +++ b/apps/daemon/internal/agent/mcode/environment_mcp_test.go @@ -32,7 +32,7 @@ func TestEnvironmentMCPUsesFixedLauncherForNewAndLoadedSessions(t *testing.T) { } t.Setenv("USER_SELECTED", "must-not-resolve-from-daemon") t.Setenv("MODEL_SECRET", "must-not-forward") - resource, err := NewExecutorFactory(&c)(t.Context(), req) + resource, err := NewExecutorFactory(testStateRoot(t), &c)(t.Context(), req) if err != nil { t.Fatal(err) } @@ -98,7 +98,7 @@ func TestEnvironmentMCPRejectsUnqualifiedAuthorityBeforePreparation(t *testing.T case "reserved": req.LocalEnvironment.MCP[0].Server.Name = "oac_workspace" } - if _, err := prepareWorkspaceOptions(c, req); err == nil || strings.Contains(err.Error(), "confidential-http-token") { + if _, err := prepareWorkspaceOptions(testStateRoot(t), c, req); err == nil || strings.Contains(err.Error(), "confidential-http-token") { t.Fatal("unqualified declaration accepted or credential exposed") } }) @@ -116,7 +116,7 @@ func TestEnvironmentMCPCancelSettlesPendingObservationBeforeDone(t *testing.T) { if err := os.WriteFile(c.Binary, []byte(strings.Replace(string(script), "HELPER=prepared", "HELPER=prepared-mcp-cancel", 1)), 0700); err != nil { t.Fatal(err) } - resource, err := NewExecutorFactory(&c)(t.Context(), req) + resource, err := NewExecutorFactory(testStateRoot(t), &c)(t.Context(), req) if err != nil { t.Fatal(err) } @@ -190,7 +190,7 @@ func TestEnvironmentHTTPMCPUsesEphemeralACPConfiguration(t *testing.T) { item.BearerToken = &value } req.LocalEnvironment.MCP = []proto.EnvironmentMCP{item} - resource, err := NewExecutorFactory(&c)(t.Context(), req) + resource, err := NewExecutorFactory(testStateRoot(t), &c)(t.Context(), req) if err != nil { t.Fatal(err) } @@ -237,7 +237,7 @@ func TestPublicEnvironmentHTTPMCPKeepsCredentialTransient(t *testing.T) { c.Network, req.LocalEnvironment.NetworkAccess = "enabled", "enabled" token := "selected-public-vault-canary" req.MCPHTTPServers = &[]proto.MCPHTTPServer{{ConnectionOrigin: "environment", ServerLabel: "remote", ServerURL: "https://example.test/mcp", BearerToken: &token}} - opts, err := prepareWorkspaceOptions(c, req) + opts, err := prepareWorkspaceOptions(testStateRoot(t), c, req) if err != nil { t.Fatal(err) } @@ -266,12 +266,12 @@ func TestPublicEnvironmentHTTPMCPKeepsCredentialTransient(t *testing.T) { } empty := []string{} (*req.MCPHTTPServers)[0].AllowedTools = &empty - if _, err := prepareWorkspaceOptions(c, req); err == nil { + if _, err := prepareWorkspaceOptions(testStateRoot(t), c, req); err == nil { t.Fatal("empty allowlist silently treated as all") } (*req.MCPHTTPServers)[0].AllowedTools = nil (*req.MCPHTTPServers)[0].Required = true - if _, err := prepareWorkspaceOptions(c, req); err == nil { + if _, err := prepareWorkspaceOptions(testStateRoot(t), c, req); err == nil { t.Fatal("required initialization silently ignored") } } diff --git a/apps/daemon/internal/agent/mcode/execution_test.go b/apps/daemon/internal/agent/mcode/execution_test.go index 9f188f97f..afa6b0ad1 100644 --- a/apps/daemon/internal/agent/mcode/execution_test.go +++ b/apps/daemon/internal/agent/mcode/execution_test.go @@ -14,7 +14,7 @@ func TestExecutionOptionsInheritUserEnvironment(t *testing.T) { r := testRequest(t) t.Setenv("OAC_TEST_SECRET_CANARY", "secret") t.Setenv("NODE_OPTIONS", "--import=untrusted") - opts, err := prepareOptions(r) + opts, err := prepareOptions(testStateRoot(t), r) if err != nil { t.Fatal(err) } @@ -48,7 +48,7 @@ func TestExecutionRejectsUnqualifiedAuthority(t *testing.T) { } { r := testRequest(t) change(&r) - if _, err := prepareOptions(r); err == nil { + if _, err := prepareOptions(testStateRoot(t), r); err == nil { t.Fatal("unsupported execution accepted") } } diff --git a/apps/daemon/internal/agent/mcode/executor.go b/apps/daemon/internal/agent/mcode/executor.go index 33ea18ed9..0ea08bf3b 100644 --- a/apps/daemon/internal/agent/mcode/executor.go +++ b/apps/daemon/internal/agent/mcode/executor.go @@ -28,7 +28,7 @@ var _ agent.Executor = (*executor)(nil) var _ agent.Turn = (*Session)(nil) // NewExecutorFactory fixes the deployment workspace once; nil selects none. -func NewExecutorFactory(config *WorkspaceConfig) agent.ExecutorFactory { +func NewExecutorFactory(runtimeRoot string, config *WorkspaceConfig) agent.ExecutorFactory { var frozen *WorkspaceConfig if config != nil { value := *config @@ -46,10 +46,10 @@ func NewExecutorFactory(config *WorkspaceConfig) agent.ExecutorFactory { var opts launchOptions binary := defaultBinary() if frozen == nil { - opts, err = prepareOptions(req) + opts, err = prepareOptions(runtimeRoot, req) } else { binary = frozen.Binary - opts, err = prepareWorkspaceOptions(*frozen, req) + opts, err = prepareWorkspaceOptions(runtimeRoot, *frozen, req) } if err != nil { return nil, err diff --git a/apps/daemon/internal/agent/mcode/executor_native_test.go b/apps/daemon/internal/agent/mcode/executor_native_test.go index d2c40d507..7f5b6afc5 100644 --- a/apps/daemon/internal/agent/mcode/executor_native_test.go +++ b/apps/daemon/internal/agent/mcode/executor_native_test.go @@ -33,7 +33,7 @@ func TestNativeMCodeExecutorReuse(t *testing.T) { if json.Unmarshal(raw, &req.AgentOptions) != nil { t.Fatal("invalid private provider options") } - value, err := NewExecutorFactory(nil)(ctx, req) + value, err := NewExecutorFactory(testStateRoot(t), nil)(ctx, req) if err != nil { t.Fatal("native Executor preparation failed") } @@ -153,7 +153,7 @@ func TestNativeMCodeExecutorReuse(t *testing.T) { } cleanupStop() req.AgentSessionID = nativeID - recovered, recoverErr := NewExecutorFactory(nil)(ctx, req) + recovered, recoverErr := NewExecutorFactory(testStateRoot(t), nil)(ctx, req) if recoverErr != nil || recovered == nil { t.Fatal("exact native history recovery failed") } diff --git a/apps/daemon/internal/agent/mcode/executor_test.go b/apps/daemon/internal/agent/mcode/executor_test.go index 59a1f5d60..d9936a712 100644 --- a/apps/daemon/internal/agent/mcode/executor_test.go +++ b/apps/daemon/internal/agent/mcode/executor_test.go @@ -45,12 +45,12 @@ func executorFixture(t *testing.T, scenario string, workspace bool) (*executor, } var factory agent.ExecutorFactory if workspace { - factory = NewExecutorFactory(&config) + factory = NewExecutorFactory(testStateRoot(t), &config) } else { req = testRequest(t) req.RunID, req.Input = "", nil t.Setenv("OAC_RUNTIME_MCODE_BIN", config.Binary) - factory = NewExecutorFactory(nil) + factory = NewExecutorFactory(testStateRoot(t), nil) } value, err := factory(t.Context(), req) if err != nil { @@ -293,7 +293,7 @@ func TestExecutorFactoryPreparationFailureHasNoTypedNilOwner(t *testing.T) { if err := os.WriteFile(config.Binary, []byte("#!/bin/sh\nexit 1\n"), 0700); err != nil { t.Fatal(err) } - value, err := NewExecutorFactory(&config)(t.Context(), req) + value, err := NewExecutorFactory(testStateRoot(t), &config)(t.Context(), req) if err == nil || value != nil { t.Fatalf("settled preparation failure returned owner: nil=%t error=%v", value == nil, err) } diff --git a/apps/daemon/internal/agent/mcode/harness_config_test.go b/apps/daemon/internal/agent/mcode/harness_config_test.go index a704912f9..1de262c5d 100644 --- a/apps/daemon/internal/agent/mcode/harness_config_test.go +++ b/apps/daemon/internal/agent/mcode/harness_config_test.go @@ -8,11 +8,11 @@ import ( func TestNativeConfigDoesNotPretendToSupportParameters(t *testing.T) { req := testRequest(t) req.AgentOptions["harness_config"] = map[string]any{} - if _, err := prepareOptions(req); err != nil { + if _, err := prepareOptions(testStateRoot(t), req); err != nil { t.Fatal(err) } req.AgentOptions["harness_config"] = map[string]any{"temperature": 0.5} - if _, err := prepareOptions(req); err != harnessconfig.ErrHarnessConfig { + if _, err := prepareOptions(testStateRoot(t), req); err != harnessconfig.ErrHarnessConfig { t.Fatalf("unsupported parameter accepted: %v", err) } } diff --git a/apps/daemon/internal/agent/mcode/model_provider_test.go b/apps/daemon/internal/agent/mcode/model_provider_test.go index f97b50508..bdef5fa6c 100644 --- a/apps/daemon/internal/agent/mcode/model_provider_test.go +++ b/apps/daemon/internal/agent/mcode/model_provider_test.go @@ -16,7 +16,7 @@ func TestOptionsModelProviderProtocols(t *testing.T) { req := testRequest(t) req.AgentOptions["model_provider"].(map[string]any)["protocol"] = tc.protocol req.AgentOptions["model"] = "chosen-model" - opts, err := prepareOptions(req) + opts, err := prepareOptions(testStateRoot(t), req) if err != nil { t.Fatal(err) } @@ -60,7 +60,7 @@ func TestOptionsRejectInvalidProvider(t *testing.T) { t.Run(tc.name, func(t *testing.T) { req := testRequest(t) req.AgentOptions["model_provider"].(map[string]any)[tc.field] = tc.value - if _, err := prepareOptions(req); err == nil { + if _, err := prepareOptions(testStateRoot(t), req); err == nil { t.Fatal("invalid model provider accepted") } }) @@ -69,7 +69,7 @@ func TestOptionsRejectInvalidProvider(t *testing.T) { req := testRequest(t) req.AgentOptions["mcode_provider"] = req.AgentOptions["model_provider"] delete(req.AgentOptions, "model_provider") - if _, err := prepareOptions(req); err == nil { + if _, err := prepareOptions(testStateRoot(t), req); err == nil { t.Fatal("retired native provider option accepted") } }) @@ -78,7 +78,7 @@ func TestOptionsRejectInvalidProvider(t *testing.T) { func TestOptionsAllowLoopbackProviderFixture(t *testing.T) { req := testRequest(t) req.AgentOptions["model_provider"].(map[string]any)["base_url"] = "http://127.0.0.1:4321/v1" - if _, err := prepareOptions(req); err != nil { + if _, err := prepareOptions(testStateRoot(t), req); err != nil { t.Fatal(err) } } diff --git a/apps/daemon/internal/agent/mcode/options.go b/apps/daemon/internal/agent/mcode/options.go index 6fc7e3492..f34f57b53 100644 --- a/apps/daemon/internal/agent/mcode/options.go +++ b/apps/daemon/internal/agent/mcode/options.go @@ -18,7 +18,7 @@ type launchOptions struct { MCP []map[string]any } -func prepareOptions(req proto.PromptRequestPayload) (launchOptions, error) { +func prepareOptions(runtimeRoot string, req proto.PromptRequestPayload) (launchOptions, error) { var result launchOptions if _, err := harnessconfiguration.Configuration().PrepareHarnessConfig(req.AgentOptions); err != nil { return result, err @@ -26,7 +26,7 @@ func prepareOptions(req proto.PromptRequestPayload) (launchOptions, error) { if err := validateExecutionRequest(req); err != nil { return result, err } - dataDir, err := agent.StateDir("mcode", req.AgentStateKey) + dataDir, err := agent.StateDir(runtimeRoot, "mcode", req.AgentStateKey) if err != nil { return result, err } diff --git a/apps/daemon/internal/agent/mcode/options_test.go b/apps/daemon/internal/agent/mcode/options_test.go index 70d736c86..5018a743f 100644 --- a/apps/daemon/internal/agent/mcode/options_test.go +++ b/apps/daemon/internal/agent/mcode/options_test.go @@ -13,7 +13,7 @@ import ( func TestOptionsRefreshManagedState(t *testing.T) { t.Setenv("MINIMAX_DATA_DIR", "/wrong") req := testRequest(t) - opts, err := prepareOptions(req) + opts, err := prepareOptions(testStateRoot(t), req) if err != nil { t.Fatal(err) } @@ -31,7 +31,7 @@ func TestOptionsRefreshManagedState(t *testing.T) { } req.AgentOptions["system_prompt"] = "" req.AgentSessionID = "native-1" - refreshed, err := prepareOptions(req) + refreshed, err := prepareOptions(testStateRoot(t), req) if err != nil { t.Fatal(err) } @@ -75,7 +75,7 @@ func TestOptionsRejectDroppedContext(t *testing.T) { t.Run(tt.name, func(t *testing.T) { req := testRequest(t) tt.edit(&req) - if _, err := prepareOptions(req); err == nil { + if _, err := prepareOptions(testStateRoot(t), req); err == nil { t.Fatal("expected validation failure") } }) diff --git a/apps/daemon/internal/agent/mcode/session_test.go b/apps/daemon/internal/agent/mcode/session_test.go index 0dfc9e070..044a1bb35 100644 --- a/apps/daemon/internal/agent/mcode/session_test.go +++ b/apps/daemon/internal/agent/mcode/session_test.go @@ -47,7 +47,7 @@ func helperRequest(t *testing.T, scenario string, resume bool) proto.PromptReque func prepareExecutor(t *testing.T, ctx context.Context, req proto.PromptRequestPayload) (*executor, error) { t.Helper() req.RunID, req.Input = "", nil - value, err := NewExecutorFactory(nil)(ctx, req) + value, err := NewExecutorFactory(testStateRoot(t), nil)(ctx, req) if value == nil { return nil, err } diff --git a/apps/daemon/internal/agent/mcode/state_root_test.go b/apps/daemon/internal/agent/mcode/state_root_test.go new file mode 100644 index 000000000..6c7919cc5 --- /dev/null +++ b/apps/daemon/internal/agent/mcode/state_root_test.go @@ -0,0 +1,16 @@ +package mcode + +import ( + "os" + "testing" +) + +func testStateRoot(t *testing.T) string { + t.Helper() + if root := os.Getenv("OAC_RUNTIME_HOME"); root != "" { + return root + } + root := t.TempDir() + t.Setenv("OAC_RUNTIME_HOME", root) + return root +} diff --git a/apps/daemon/internal/agent/mcode/tool_environment_test.go b/apps/daemon/internal/agent/mcode/tool_environment_test.go index ab6accd5a..fd047583a 100644 --- a/apps/daemon/internal/agent/mcode/tool_environment_test.go +++ b/apps/daemon/internal/agent/mcode/tool_environment_test.go @@ -60,7 +60,7 @@ func TestWorkspaceCredentialsRemainInRuntimeSnapshotAcrossReconnect(t *testing.T if err != nil { t.Fatal(err) } - opts, err := prepareWorkspaceOptions(config, prepared) + opts, err := prepareWorkspaceOptions(testStateRoot(t), config, prepared) if err != nil { t.Fatal(err) } @@ -125,7 +125,7 @@ func TestWorkspaceRejectsInvalidToolEnvironment(t *testing.T) { } t.Setenv("OAC_RUNTIME_TOOL_ENV_FILE", file) t.Setenv("OAC_RUNTIME_INITIALIZATION_DIRECTORY", t.TempDir()) - if _, err := prepareWorkspaceOptions(config, req); err == nil { + if _, err := prepareWorkspaceOptions(testStateRoot(t), config, req); err == nil { t.Fatal("invalid explicit tool configuration was ignored") } } diff --git a/apps/daemon/internal/agent/mcode/workspace.go b/apps/daemon/internal/agent/mcode/workspace.go index 5b913e4ac..b5bbd2899 100644 --- a/apps/daemon/internal/agent/mcode/workspace.go +++ b/apps/daemon/internal/agent/mcode/workspace.go @@ -49,7 +49,7 @@ func ConfigureLocal(binary, node, bridge, root, workspace string, network agentn return c, nil } -func prepareWorkspaceOptions(c WorkspaceConfig, req proto.PromptRequestPayload) (launchOptions, error) { +func prepareWorkspaceOptions(runtimeRoot string, c WorkspaceConfig, req proto.PromptRequestPayload) (launchOptions, error) { if c.Network != "enabled" || len(c.AllowedDomains) != 0 { return launchOptions{}, fmt.Errorf("mcode: Runtime does not implement network isolation") } @@ -66,7 +66,7 @@ func prepareWorkspaceOptions(c WorkspaceConfig, req proto.PromptRequestPayload) private.LocalEnvironment, private.DisableExecutionEnvironment = nil, true // Public declarations have already been resolved into the transient ACP map. private.MCPHTTPServers = nil - opts, err := prepareOptions(private) + opts, err := prepareOptions(runtimeRoot, private) if err != nil { return opts, err } diff --git a/apps/daemon/internal/agent/mcode/workspace_network_test.go b/apps/daemon/internal/agent/mcode/workspace_network_test.go index 83b3156ee..57aed9044 100644 --- a/apps/daemon/internal/agent/mcode/workspace_network_test.go +++ b/apps/daemon/internal/agent/mcode/workspace_network_test.go @@ -7,7 +7,7 @@ func TestWorkspaceRejectsInnerNetworkIsolation(t *testing.T) { c, req, _ := workspaceFixture(t) c.Network = access req.LocalEnvironment.NetworkAccess = access - if _, err := prepareWorkspaceOptions(c, req); err == nil { + if _, err := prepareWorkspaceOptions(testStateRoot(t), c, req); err == nil { t.Fatal("unsupported network isolation accepted") } } diff --git a/apps/daemon/internal/agent/mcode/workspace_skills_test.go b/apps/daemon/internal/agent/mcode/workspace_skills_test.go index 0d7c9470d..1126eaac9 100644 --- a/apps/daemon/internal/agent/mcode/workspace_skills_test.go +++ b/apps/daemon/internal/agent/mcode/workspace_skills_test.go @@ -28,7 +28,7 @@ func TestWorkspaceSkillsUseSelectedSnapshotAndNativeLoader(t *testing.T) { Metadata: agentskill.Metadata{Name: "proof", Description: "Read a marker"}, }} for range 2 { - opts, err := prepareWorkspaceOptions(c, req) + opts, err := prepareWorkspaceOptions(testStateRoot(t), c, req) if err != nil { t.Fatal(err) } diff --git a/apps/daemon/internal/agent/runtime_paths.go b/apps/daemon/internal/agent/runtime_paths.go index 456231282..ce61e9a89 100644 --- a/apps/daemon/internal/agent/runtime_paths.go +++ b/apps/daemon/internal/agent/runtime_paths.go @@ -4,16 +4,13 @@ import ( "fmt" "path/filepath" "strings" - - "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/paths" ) // StateDir returns an adapter-owned state directory scoped to one agent // state. It never derives runtime state from the subprocess cwd. -func StateDir(agentKind, agentStateKey string) (string, error) { - root, err := paths.Root() - if err != nil { - return "", fmt.Errorf("agent: resolve state directory: %w", err) +func StateDir(root, agentKind, agentStateKey string) (string, error) { + if !filepath.IsAbs(root) || filepath.Clean(root) != root { + return "", fmt.Errorf("agent: state root must be a clean absolute directory") } kind := safeRuntimePathPart(agentKind) if kind == "" { diff --git a/apps/daemon/internal/agent/runtime_paths_test.go b/apps/daemon/internal/agent/runtime_paths_test.go index 3e98c1792..7abd9ebd8 100644 --- a/apps/daemon/internal/agent/runtime_paths_test.go +++ b/apps/daemon/internal/agent/runtime_paths_test.go @@ -8,7 +8,7 @@ import ( func TestStateDirUsesStableAgentState(t *testing.T) { home := t.TempDir() t.Setenv("OAC_RUNTIME_HOME", home) - got, err := StateDir("codex", "conv-1/agent-1/codex") + got, err := StateDir(home, "codex", "conv-1/agent-1/codex") if err != nil { t.Fatalf("StateDir: %v", err) } @@ -22,11 +22,11 @@ func TestStateDirRequiresAgentState(t *testing.T) { home := t.TempDir() t.Setenv("OAC_RUNTIME_HOME", home) for _, key := range []string{"", " ", "../.."} { - if _, err := StateDir("codex", key); err == nil { + if _, err := StateDir(home, "codex", key); err == nil { t.Fatalf("state key %q accepted", key) } } - got, err := StateDir("codex", "../session-1/a b") + got, err := StateDir(home, "codex", "../session-1/a b") if want := filepath.Join(home, "runtime", "codex", "state", "session-1", "a_b"); err != nil || got != want { t.Fatalf("sanitized state = %q, %v; want %q", got, err, want) } diff --git a/apps/daemon/internal/cli/agent_discovery.go b/apps/daemon/internal/cli/agent_discovery.go index 0b96f6c34..b61818544 100644 --- a/apps/daemon/internal/cli/agent_discovery.go +++ b/apps/daemon/internal/cli/agent_discovery.go @@ -9,6 +9,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/agent/claudesdk" "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/agent/codex" "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/agent/mcode" + "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/paths" obslog "github.com/MiniMax-AI/OpenAgentCore/internal/obs/log" ) @@ -24,6 +25,14 @@ func preflightAgentCLIs(parent context.Context, rc *runContext, profile string) return discoverAgentCLIs(parent, rc, profile, harnessDeclarations) } func discoverAgentCLIs(parent context.Context, rc *runContext, profile string, declarations []agent.Declaration) (agentCLIDiscovery, error) { + runtimeRoot, err := paths.Root() + if err != nil { + return nil, err + } + stateRoot, err := paths.StateDirectory() + if err != nil { + return nil, err + } var out agentCLIDiscovery available := false for _, declaration := range declarations { @@ -31,7 +40,7 @@ func discoverAgentCLIs(parent context.Context, rc *runContext, profile string, d continue } started := time.Now() - runtime := declaration.Discover(parent, agent.DiscoveryOptions{Profile: profile, Stdout: rc.stdout, Stderr: rc.stderr}, declaration.Info) + runtime := declaration.Discover(parent, agent.DiscoveryOptions{Profile: profile, StateRoot: stateRoot, RuntimeRoot: runtimeRoot, Stdout: rc.stdout, Stderr: rc.stderr}, declaration.Info) obslog.Ctx(parent).Info("runtime startup stage", "stage", "harness_discovery", "harness_kind", declaration.Info.Kind, "duration_ms", float64(time.Since(started))/float64(time.Millisecond), "available", runtime != nil && runtime.Info.Available) diff --git a/apps/daemon/internal/cli/connect.go b/apps/daemon/internal/cli/connect.go index 43583ce79..b6ab804b2 100644 --- a/apps/daemon/internal/cli/connect.go +++ b/apps/daemon/internal/cli/connect.go @@ -99,10 +99,6 @@ func runConnect(ctx *runContext, args []string) error { } initializeRuntimeObservations(ctx) - discover := func(parent context.Context) (agentCLIDiscovery, error) { - return preflightAgentCLIs(parent, ctx, *profile) - } - var prof auth.Profile if bootstrapped != nil { prof = *bootstrapped @@ -113,7 +109,7 @@ func runConnect(ctx *runContext, args []string) error { } } - return mainLoopRemoteWithDiscovery(context.Background(), ctx, *profile, prof, "", discover) + return mainLoop(ctx, *profile, prof) } // spawnBackground forks the daemon into the background. Parent @@ -193,7 +189,6 @@ func mainLoopRemote(parent context.Context, rc *runContext, profile string, prof return preflightAgentCLIs(ctx, rc, profile) }) } - func mainLoopRemoteWithDiscovery(parent context.Context, rc *runContext, profile string, prof auth.Profile, remote string, discover func(context.Context) (agentCLIDiscovery, error)) error { // Route through obs/log so daemon log lines pick up the same // trace_id / span_id auto-injection as the server side — when the @@ -209,20 +204,20 @@ func mainLoopRemoteWithDiscovery(parent context.Context, rc *runContext, profile rootCtx, cancel := daemonize.NotifyContext(parent) defer cancel() - boot, agentCLIs, err := prepareConnection(rootCtx, discover, - func(ctx context.Context) (boot *transport.BootstrapResponse, err error) { - started := time.Now() - defer func() { observeRuntimeStartup(ctx, "bootstrap", started, err) }() - bootCtx, stop := context.WithTimeout(ctx, bootstrapTimeout) - defer stop() - if remote != "" { - return environmentBootstrap(bootCtx, prof, remote) - } - return transport.Bootstrap(bootCtx, prof.ServerURL, prof.RuntimeID, prof.RunnerCredential, Version) - }) + boot, agentCLIs, err := prepareConnection(rootCtx, discover, func(ctx context.Context) (boot *transport.BootstrapResponse, err error) { + started := time.Now() + defer func() { observeRuntimeStartup(ctx, "bootstrap", started, err) }() + bootCtx, stop := context.WithTimeout(ctx, bootstrapTimeout) + defer stop() + if remote != "" { + return environmentBootstrap(bootCtx, prof, remote) + } + return transport.Bootstrap(bootCtx, prof.ServerURL, prof.RuntimeID, prof.RunnerCredential, Version) + }) if err != nil { return err } + wsURL, err := transport.DeriveWSURL(*boot, prof.ServerURL) if err != nil { return fmt.Errorf("connect: derive ws url: %w", err) diff --git a/apps/daemon/internal/cli/connect_suspend.go b/apps/daemon/internal/cli/connect_suspend.go index b1146b94d..3d3228ff1 100644 --- a/apps/daemon/internal/cli/connect_suspend.go +++ b/apps/daemon/internal/cli/connect_suspend.go @@ -201,7 +201,7 @@ func (s *suspendedRouter) pump(ctx context.Context, conn *transport.Conn, boot * if sendErr := sendSuspendResult(ctx, conn, env, proto.TypeEnvironmentQuiesced, request, "resource_busy"); sendErr != nil { return nil, sendErr } - if errors.Is(err, context.DeadlineExceeded) || errors.Is(err, context.Canceled) { + if errors.Is(err, dispatch.ErrRouterQuiesced) || errors.Is(err, dispatch.ErrRouterClosed) || errors.Is(err, context.DeadlineExceeded) || errors.Is(err, context.Canceled) { return nil, err } continue diff --git a/apps/daemon/internal/cli/connect_suspend_test.go b/apps/daemon/internal/cli/connect_suspend_test.go index 4c7177746..6d92648e2 100644 --- a/apps/daemon/internal/cli/connect_suspend_test.go +++ b/apps/daemon/internal/cli/connect_suspend_test.go @@ -17,7 +17,11 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/agent" "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/transport" + "github.com/MiniMax-AI/OpenAgentCore/internal/agentcapabilities" "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" + "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto/prototest" + "github.com/MiniMax-AI/OpenAgentCore/internal/harnessconfig" + "github.com/google/uuid" "github.com/gorilla/websocket" ) @@ -334,3 +338,156 @@ func TestSuspensionReconnectBeforeConfirmation(t *testing.T) { }) } } + +func TestQuiesceRetainsExecutorAcrossReconnectAndShutdownConfirmsCleanup(t *testing.T) { + environment, session := uuid.NewString(), uuid.NewString() + t.Setenv("OAC_RUNTIME_HOME", t.TempDir()) + t.Setenv("OAC_RUNTIME_STATE_DIRECTORY", t.TempDir()) + t.Setenv("OAC_RUNTIME_WORKSPACE", t.TempDir()) + t.Setenv("OAC_RUNTIME_CAPABILITY_DIRECTORY", t.TempDir()) + t.Setenv("OAC_RUNTIME_ENVIRONMENT_ID", environment) + t.Setenv("OAC_RUNTIME_SESSION_ID", session) + t.Setenv("OAC_RUNTIME_NETWORK_ACCESS", "disabled") + t.Setenv("OAC_RUNTIME_ALLOWED_DOMAINS", "") + peers := make(chan *websocket.Conn, 2) + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + peer, err := (&websocket.Upgrader{}).Upgrade(w, r, nil) + if err == nil { + peers <- peer + } + })) + defer server.Close() + ctx, cancel := context.WithCancel(t.Context()) + defer cancel() + dial := func(ctx context.Context) (*transport.Conn, error) { + return transport.Dial(ctx, transport.DialOptions{WSURL: "ws" + strings.TrimPrefix(server.URL, "http"), DeviceID: "device", Credential: "fixture", DaemonVersion: proto.Version}) + } + native := &cleanupExecutor{retry: make(chan struct{}), confirm: make(chan struct{})} + defer func() { + select { + case <-native.confirm: + default: + close(native.confirm) + } + }() + registry := agent.NewRegistry() + registry.RegisterKind(proto.SupportedAgentKind{Kind: "cleanup", Available: true, Capabilities: prototest.Capabilities(proto.AgentKindCapabilities{LocalEnvironment: proto.CapabilitySupported})}, harnessconfig.Configuration{}) + registry.RegisterExecutor("cleanup", func(context.Context, proto.PromptRequestPayload) (agent.Executor, error) { return native, nil }) + control := &suspendControl{path: filepath.Join(t.TempDir(), "control.json"), identity: suspendIdentity{EnvironmentID: environment}, signal: make(chan os.Signal, 1)} + finished := make(chan error, 1) + go func() { + finished <- runSuspendLoop(ctx, dial, registry, &transport.BootstrapResponse{HeartbeatSeconds: 60}, agentCLIDiscovery{}, control) + }() + var peer *websocket.Conn + select { + case peer = <-peers: + case <-time.After(3 * time.Second): + t.Fatal("no connection") + } + defer peer.Close() + preparation, err := proto.NewEnvelope(proto.TypeExecutionPrepare, "prepare", proto.ExecutionPreparePayload{SessionID: session, Configuration: proto.PromptRequestPayload{AgentKind: "cleanup", AgentStateKey: "agents-api-" + session, LocalEnvironment: &proto.LocalEnvironment{ID: environment, NetworkAccess: "disabled", WorkspaceDirectory: "/workspace", CapabilitySources: &agentcapabilities.Input{}}}}) + if err != nil { + t.Fatal(err) + } + if err = peer.WriteJSON(preparation); err != nil { + t.Fatal(err) + } + readStatus := func(want string) proto.PreparationStatusPayload { + t.Helper() + _ = peer.SetReadDeadline(time.Now().Add(3 * time.Second)) + for { + var frame proto.Envelope + if err := peer.ReadJSON(&frame); err != nil { + t.Fatal(err) + } + if frame.Type != proto.TypePreparationStatus { + continue + } + var status proto.PreparationStatusPayload + if err := frame.DecodePayload(&status); err != nil { + t.Fatal(err) + } + if status.State == want { + return status + } + if status.State == "failed" || status.State == "rejected" { + t.Fatalf("preparation failed: %+v", status) + } + } + } + ready := readStatus("ready") + request := proto.EnvironmentSuspendPayload{EnvironmentID: environment, SuspendID: "attempt"} + // Admission is still held: ordinary busy must leave the socket usable. + sendLifecycleFrame(t, peer, proto.TypeEnvironmentQuiesce, request) + if got := readLifecycleResult(t, peer, proto.TypeEnvironmentQuiesced); got.Accepted || got.ErrorCode != "resource_busy" { + t.Fatalf("busy result: %+v", got) + } + release, err := proto.NewEnvelope(proto.TypeExecutionRelease, "prepare", proto.ExecutionReleasePayload{Handle: ready.Handle}) + if err != nil { + t.Fatal(err) + } + if err = peer.WriteJSON(release); err != nil { + t.Fatal("busy rejection disconnected", err) + } + readStatus("released") + sendLifecycleFrame(t, peer, proto.TypeEnvironmentQuiesce, request) + if got := readLifecycleResult(t, peer, proto.TypeEnvironmentQuiesced); !got.Accepted { + t.Fatalf("idle owner rejected: %+v", got) + } + if native.closes.Load() != 0 { + t.Fatal("quiesce closed the native owner") + } + control.signal <- syscall.SIGUSR1 + select { + case peer = <-peers: + case <-time.After(3 * time.Second): + t.Fatal("no planned reconnect") + } + defer peer.Close() + sendLifecycleFrame(t, peer, proto.TypeEnvironmentResume, request) + if got := readLifecycleResult(t, peer, proto.TypeEnvironmentResumed); !got.Accepted { + t.Fatalf("resume rejected: %+v", got) + } + preparation.ID = "resumed-prepare" + if err := peer.WriteJSON(preparation); err != nil { + t.Fatal(err) + } + resumed := readStatus("ready") + if !resumed.Reused || resumed.ExecutorID != ready.ExecutorID || resumed.Handle == ready.Handle || native.closes.Load() != 0 { + t.Fatalf("resume replaced native owner: before=%+v after=%+v closes=%d", ready, resumed, native.closes.Load()) + } + // Cancelling the daemon still retires resident native resources. Failed + // cleanup must settle before the loop can finish or create another router. + cancel() + select { + case <-native.retry: + case <-time.After(3 * time.Second): + t.Fatal("cancelled resident owner did not enter Shutdown retry") + } + _ = peer.SetReadDeadline(time.Now().Add(time.Second)) + for { + var frame proto.Envelope + if peer.ReadJSON(&frame) != nil { + break + } + } + select { + case other := <-peers: + _ = other.Close() + t.Fatal("reconnected before native cleanup settled") + default: + } + cancel() + close(native.confirm) + select { + case <-finished: + case <-time.After(3 * time.Second): + t.Fatal("confirmed cleanup failed to settle loop") + } + if native.closes.Load() != 2 { + t.Fatalf("Close calls=%d", native.closes.Load()) + } + if control.identity.SuspendID != "" || control.lastResumed == nil || !control.lastResumed.SameSuspension(request) { + t.Fatal("resumed suspension left snapshot control armed or lost its receipt") + } +} diff --git a/apps/daemon/internal/cli/native_discovery_test.go b/apps/daemon/internal/cli/native_discovery_test.go index 29776bcde..e1dd6f225 100644 --- a/apps/daemon/internal/cli/native_discovery_test.go +++ b/apps/daemon/internal/cli/native_discovery_test.go @@ -14,10 +14,16 @@ import ( func TestDiscoveryAndRegistration(t *testing.T) { for _, selected := range []string{"codex", "mcode", "claude_sdk", ""} { t.Run(selected, func(t *testing.T) { + runtimeRoot, stateRoot := t.TempDir(), t.TempDir() + t.Setenv("OAC_RUNTIME_HOME", runtimeRoot) + t.Setenv("OAC_RUNTIME_STATE_DIRECTORY", stateRoot) called := []string{} declarations := append([]agent.Declaration(nil), harnessDeclarations...) for i := range declarations { - declarations[i].Discover = func(_ context.Context, _ agent.DiscoveryOptions, info proto.SupportedAgentKind) *agent.Runtime { + declarations[i].Discover = func(_ context.Context, options agent.DiscoveryOptions, info proto.SupportedAgentKind) *agent.Runtime { + if options.RuntimeRoot != runtimeRoot || options.StateRoot != stateRoot { + t.Fatal("discovery lost explicit directory context") + } called = append(called, info.Kind) info.Available = true info.Version = "test" diff --git a/apps/daemon/internal/dispatch/executor.go b/apps/daemon/internal/dispatch/executor.go index 9c18e9ee2..5b2e94305 100644 --- a/apps/daemon/internal/dispatch/executor.go +++ b/apps/daemon/internal/dispatch/executor.go @@ -255,7 +255,7 @@ func (r *Router) abandonExecutorAdmission(p *preparationState, state, code strin } func (r *Router) scheduleExecutorIdleLocked(owner *executorState) { - if owner.invalid || owner.preparing || owner.run != nil || owner.admission != nil { + if r.closed || r.suspension != nil || owner.invalid || owner.preparing || owner.run != nil || owner.admission != nil { return } if owner.timer != nil { @@ -283,7 +283,7 @@ func (r *Router) scheduleExecutorIdleLocked(owner *executorState) { // expireIdleExecutor closes owner only while lease is its current idle lease. func (r *Router) expireIdleExecutor(owner *executorState, lease uint64) { r.mu.Lock() - if r.executors[owner.sessionID] != owner || owner.idleLease != lease || owner.run != nil || owner.admission != nil || owner.invalid { + if r.closed || r.suspension != nil || r.executors[owner.sessionID] != owner || owner.idleLease != lease || owner.run != nil || owner.admission != nil || owner.invalid { r.mu.Unlock() return } diff --git a/apps/daemon/internal/dispatch/executor_test.go b/apps/daemon/internal/dispatch/executor_test.go index a2324a78c..2cba36047 100644 --- a/apps/daemon/internal/dispatch/executor_test.go +++ b/apps/daemon/internal/dispatch/executor_test.go @@ -342,3 +342,52 @@ func TestExecutorRejectsOutputFromAnotherTurn(t *testing.T) { func (*reusableTurn) SteerWithReceipt(context.Context, proto.PromptSteerPayload, func()) error { return agent.ErrSteeringRejected } + +func TestExecutorResumesSameOwnerAndNextTurnAfterSuspension(t *testing.T) { + owner := &reusableExecutor{starts: make(chan *reusableTurn, 2)} + var factories atomic.Int32 + registry := agent.NewRegistry() + registerExecutorKind(registry, proto.SupportedAgentKind{Kind: "prepared", Available: true, Capabilities: prototest.Capabilities(proto.AgentKindCapabilities{LocalEnvironment: proto.CapabilitySupported, DurableInputReceipts: proto.CapabilitySupported})}, func(context.Context, proto.PromptRequestPayload) (agent.Executor, error) { + factories.Add(1) + return owner, nil + }) + sender := &recSender{} + r, err := dispatch.New(dispatch.Config{Registry: registry, Sender: sender, LocalWorkspace: preparationWorkspace(t)}) + if err != nil { + t.Fatal(err) + } + t.Cleanup(func() { + if err := r.Shutdown(context.Background()); err != nil { + t.Error(err) + } + }) + first := executorAdmission(t, r, sender, "first", preparationRequest()) + startExecutorTurn(t, r, sender, "first", "one", first) + (<-owner.starts).finish() + waitFor(t, func() bool { return r.ActiveRuns() == 0 }, "first Turn settled") + request := proto.EnvironmentSuspendPayload{EnvironmentID: preparationEnvironmentID, SuspendID: "suspension"} + if err := r.Quiesce(t.Context(), request); err != nil { + t.Fatal(err) + } + resumedSender := &recSender{} + if err := r.Resume(request, resumedSender); err != nil { + t.Fatal(err) + } + req := preparationRequest() + req.Configuration.AgentSessionID = "native-session" + conflict := req + conflict.Configuration.AgentOptions = map[string]any{"model": "changed"} + if err := r.Handle(t.Context(), mustEnv(t, proto.TypeExecutionPrepare, "conflict", conflict)); err == nil { + t.Fatal("suspension relaxed the fixed configuration") + } + second := executorAdmission(t, r, resumedSender, "second", req) + if !second.Reused || first.ExecutorID != second.ExecutorID || first.Handle == second.Handle || factories.Load() != 1 || owner.closes.Load() != 0 { + t.Fatal("suspension replaced the resident native owner") + } + startExecutorTurn(t, r, resumedSender, "second", "two", second) + (<-owner.starts).finish() + waitFor(t, func() bool { return r.ActiveRuns() == 0 }, "resumed Turn settled") + if owner.closes.Load() != 0 { + t.Fatal("resumed Turn retired its healthy native owner") + } +} diff --git a/apps/daemon/internal/dispatch/functions_native_test.go b/apps/daemon/internal/dispatch/functions_native_test.go index f8fe3879b..b022d94dd 100644 --- a/apps/daemon/internal/dispatch/functions_native_test.go +++ b/apps/daemon/internal/dispatch/functions_native_test.go @@ -108,7 +108,7 @@ func TestNativeFunctionBridge(t *testing.T) { })) defer model.Close() reg := agent.NewRegistry() - registerExecutorKind(reg, proto.SupportedAgentKind{Kind: "codex", Available: true, Capabilities: prototest.Capabilities(proto.AgentKindCapabilities{FunctionTools: proto.CapabilitySupported, EnvironmentNone: proto.CapabilitySupported})}, codex.NewExecutorFactory()) + registerExecutorKind(reg, proto.SupportedAgentKind{Kind: "codex", Available: true, Capabilities: prototest.Capabilities(proto.AgentKindCapabilities{FunctionTools: proto.CapabilitySupported, EnvironmentNone: proto.CapabilitySupported})}, codex.NewExecutorFactory(t.TempDir())) sender := make(nativeFunctionSender, 256) ctx, cancel := context.WithTimeout(t.Context(), 60*time.Second) defer cancel() diff --git a/apps/daemon/internal/dispatch/runtime_preparation.go b/apps/daemon/internal/dispatch/runtime_preparation.go index 1c777f0fe..926ddc5f9 100644 --- a/apps/daemon/internal/dispatch/runtime_preparation.go +++ b/apps/daemon/internal/dispatch/runtime_preparation.go @@ -13,19 +13,18 @@ import ( "github.com/google/uuid" ) -const runtimePreparationTimeout = 120 * time.Second - // Router.mu protects one connection-local transfer. Partial installation data // belongs to the bound Environment and is never removed by transfer cleanup. type runtimePreparationTransfer struct { - envelope proto.Envelope - request proto.RuntimePreparePayload - data []byte - ready chan struct{} - cancel context.CancelFunc - finished bool - apply bool - uncertain bool + transferDeadline time.Time + envelope proto.Envelope + request proto.RuntimePreparePayload + data []byte + ready chan struct{} + cancel context.CancelFunc + finished bool + apply bool + uncertain bool } func (r *Router) handleRuntimePrepare(ctx context.Context, env proto.Envelope) error { @@ -65,9 +64,10 @@ func (r *Router) handleRuntimePrepare(ctx context.Context, env proto.Envelope) e r.mu.Unlock() return r.sendRuntimePrepareResult(ctx, env.ID, rejectedRuntimePreparation("resource_unavailable")) } - owner, cancel := context.WithTimeout(context.WithoutCancel(ctx), runtimePreparationTimeout) + owner, cancel := context.WithTimeout(context.WithoutCancel(ctx), time.Duration(request.BudgetMS)*time.Millisecond) u := &runtimePreparationTransfer{ - envelope: env, request: request, data: make([]byte, 0, request.SizeBytes), + transferDeadline: time.Now().Add(time.Duration(proto.RuntimePrepareTransferBudgetMS) * time.Millisecond), + envelope: env, request: request, data: make([]byte, 0, request.SizeBytes), ready: make(chan struct{}), cancel: cancel, } r.runtimePreparation = u @@ -100,7 +100,7 @@ func (r *Router) handleRuntimePrepare(ctx context.Context, env proto.Envelope) e return nil } apply := false - if request.Step == "commit" && len(u.data) == u.request.SizeBytes { + if request.Step == "commit" && len(u.data) == u.request.SizeBytes && time.Now().Before(u.transferDeadline) { if u.request.Action == "finalize" || u.request.Action == "initialize" { apply = true } else { @@ -135,7 +135,10 @@ func (r *Router) finishRuntimePreparationTransferLocked(u *runtimePreparationTra func (r *Router) runRuntimePreparationTransfer(ctx context.Context, u *runtimePreparationTransfer, apply func(context.Context, proto.RuntimePreparePayload, []byte) error) { defer r.shutdownWG.Done() defer u.cancel() + staging := time.NewTimer(time.Until(u.transferDeadline)) + defer staging.Stop() select { + case <-staging.C: case <-u.ready: case <-r.shutdownCh: case <-ctx.Done(): diff --git a/apps/daemon/internal/dispatch/runtime_preparation_test.go b/apps/daemon/internal/dispatch/runtime_preparation_test.go index 7a8a31c70..ba8631a86 100644 --- a/apps/daemon/internal/dispatch/runtime_preparation_test.go +++ b/apps/daemon/internal/dispatch/runtime_preparation_test.go @@ -9,6 +9,7 @@ import ( "os" "path/filepath" "testing" + "testing/synctest" "time" "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/agent" @@ -57,7 +58,7 @@ func capabilityEnvelope(t *testing.T, id string, request proto.RuntimePreparePay func capabilityBegin(environment, session string, body []byte) proto.RuntimePreparePayload { digest := sha256.Sum256(body) return proto.RuntimePreparePayload{ - Step: "begin", Action: "skill", EnvironmentID: environment, SessionID: session, + BudgetMS: 300000, Step: "begin", Action: "skill", EnvironmentID: environment, SessionID: session, Skill: &agentskill.Metadata{Type: "inline", Name: "proof", Description: "A proof."}, SizeBytes: len(body), SHA256: hex.EncodeToString(digest[:]), } @@ -245,9 +246,10 @@ func TestRuntimePreparationUploadBlocksWorkspaceWriteAndSuspension(t *testing.T) func TestRuntimePreparationCancellationKeepsOwnershipUntilApplyStops(t *testing.T) { r, sender, environment, session := capabilitiesTestRouter(t) + sender.frames = make(chan proto.Envelope) ctx, cancel := context.WithCancel(context.Background()) id := uuid.NewString() - request := proto.RuntimePreparePayload{Step: "begin", Action: "finalize", EnvironmentID: environment, SessionID: session, Sources: &agentcapabilities.Input{}} + request := proto.RuntimePreparePayload{BudgetMS: 300000, Step: "begin", Action: "finalize", EnvironmentID: environment, SessionID: session, Sources: &agentcapabilities.Input{}} owner := &runtimePreparationTransfer{envelope: capabilityEnvelope(t, id, request), request: request, ready: make(chan struct{}), cancel: cancel, finished: true, apply: true} close(owner.ready) r.runtimePreparation = owner @@ -262,7 +264,10 @@ func TestRuntimePreparationCancellationKeepsOwnershipUntilApplyStops(t *testing. <-ctx.Done() close(interrupted) <-release - return os.WriteFile(retained, []byte("retained"), 0400) + if err := os.WriteFile(retained, []byte("retained"), 0400); err != nil { + return err + } + return context.Canceled }) <-started wait, stop := context.WithTimeout(context.Background(), 20*time.Millisecond) @@ -279,13 +284,22 @@ func TestRuntimePreparationCancellationKeepsOwnershipUntilApplyStops(t *testing. t.Fatal("cancel released unsettled capability ownership") } close(release) + wait, stop = context.WithTimeout(context.Background(), 20*time.Millisecond) + err = r.Shutdown(wait) + stop() + if !errors.Is(err, context.DeadlineExceeded) { + t.Fatalf("shutdown abandoned pending result delivery: %v", err) + } + capabilitiesReceipt(t, sender, id, "unknown") shutdownCapabilitiesRouter(t, r) - capabilitiesReceipt(t, sender, id, "completed") if _, err := os.Stat(retained); err != nil { t.Fatal("shutdown deleted installation result") } - if r.runtimePreparation != nil { - t.Fatal("confirmed completion retained capacity") + if r.runtimePreparation != owner || !owner.uncertain { + t.Fatal("shutdown erased unknown outcome") + } + if err := r.Handle(t.Context(), capabilityEnvelope(t, uuid.NewString(), request)); !errors.Is(err, ErrRouterClosed) { + t.Fatalf("closed router admitted a successor: %v", err) } } @@ -340,9 +354,91 @@ func TestRuntimePreparationResultCategoriesAndUnknownOwnership(t *testing.T) { if !owned { t.Fatal("unknown mutation released its ownership") } - wait, stop := context.WithTimeout(context.Background(), time.Second) - defer stop() - if err := r.Shutdown(wait); err == nil { - t.Fatal("shutdown claimed uncertain mutation settled") + next := uuid.NewString() + if err := r.Handle(t.Context(), capabilityEnvelope(t, next, request)); err != nil { + t.Fatal(err) + } + if got := capabilitiesReceipt(t, sender, next, "rejected"); got.ErrorCode != "runtime_preparation_capacity" { + t.Fatalf("unknown operation lost its admission fence: %+v", got) + } + shutdownCapabilitiesRouter(t, r) + if r.runtimePreparation != owner || !owner.uncertain { + t.Fatal("shutdown erased unknown outcome") } } + +func TestRuntimePreparationStagingDeadlineRejectsWithoutApplying(t *testing.T) { + synctest.Test(t, func(t *testing.T) { + r, sender, environment, session := capabilitiesTestRouter(t) + id := uuid.NewString() + request := capabilityBegin(environment, session, []byte("abc")) + if err := r.Handle(t.Context(), capabilityEnvelope(t, id, request)); err != nil { + t.Fatal(err) + } + capabilitiesReceipt(t, sender, id, "ready") + time.Sleep(121 * time.Second) + synctest.Wait() + capabilitiesReceipt(t, sender, id, "rejected") + r.mu.Lock() + pending := r.runtimePreparation != nil + r.mu.Unlock() + if pending { + t.Fatal("uncommitted staging retained ownership") + } + shutdownCapabilitiesRouter(t, r) + }) +} + +func TestRuntimePreparationApplyOutlivesStagingAndHonorsBudget(t *testing.T) { + synctest.Test(t, func(t *testing.T) { + r, sender, environment, session := capabilitiesTestRouter(t) + ctx, cancel := context.WithTimeout(t.Context(), 5*time.Minute) + id := uuid.NewString() + request := capabilityBegin(environment, session, []byte("abc")) + owner := &runtimePreparationTransfer{envelope: capabilityEnvelope(t, id, request), request: request, ready: make(chan struct{}), cancel: cancel, finished: true, apply: true, transferDeadline: time.Now().Add(2 * time.Minute)} + close(owner.ready) + r.runtimePreparation = owner + r.shutdownWG.Add(1) + stopped := make(chan struct{}) + go r.runRuntimePreparationTransfer(ctx, owner, func(ctx context.Context, _ proto.RuntimePreparePayload, _ []byte) error { + <-ctx.Done() + close(stopped) + return ctx.Err() + }) + time.Sleep(121 * time.Second) + synctest.Wait() + select { + case <-stopped: + t.Fatal("staging timeout killed apply") + default: + } + time.Sleep(180 * time.Second) + synctest.Wait() + <-stopped + capabilitiesReceipt(t, sender, id, "unknown") + // Unknown effects remain retained; this test must not claim safe replay. + r.mu.Lock() + unknown := r.runtimePreparation == owner && owner.uncertain + r.mu.Unlock() + if !unknown { + t.Fatal("unknown apply lost ownership") + } + }) +} + +func TestRuntimePreparationHonorsBeginBudgetDuringStaging(t *testing.T) { + synctest.Test(t, func(t *testing.T) { + r, sender, environment, session := capabilitiesTestRouter(t) + id := uuid.NewString() + request := capabilityBegin(environment, session, []byte("abc")) + request.BudgetMS = 10 + if err := r.Handle(t.Context(), capabilityEnvelope(t, id, request)); err != nil { + t.Fatal(err) + } + capabilitiesReceipt(t, sender, id, "ready") + time.Sleep(11 * time.Millisecond) + synctest.Wait() + capabilitiesReceipt(t, sender, id, "rejected") + shutdownCapabilitiesRouter(t, r) + }) +} diff --git a/apps/daemon/internal/dispatch/shutdown.go b/apps/daemon/internal/dispatch/shutdown.go index b09de4c70..ec476d1f1 100644 --- a/apps/daemon/internal/dispatch/shutdown.go +++ b/apps/daemon/internal/dispatch/shutdown.go @@ -69,9 +69,9 @@ func (r *Router) runShutdownAttempt(attempt *shutdownAttempt, victims []sessionC r.shutdownWG.Wait() r.mu.Lock() - if r.runtimePreparation != nil && r.runtimePreparation.uncertain { - attempt.err = errors.Join(attempt.err, errors.New("dispatch: capability preparation remains uncertain")) - } + // Runtime preparation joins only after ApplyRuntimePreparation stops local + // mutations and its receipt send finishes. An unknown result still fences + // this closed Router, but is not outstanding cleanup. Core owns no-replay. if r.workspaceWrite != nil && r.workspaceWrite.uncertain { attempt.err = errors.Join(attempt.err, errors.New("dispatch: local workspace write remains uncertain")) } diff --git a/apps/daemon/internal/dispatch/suspend.go b/apps/daemon/internal/dispatch/suspend.go index 97a8cc6fc..b38eb9bdc 100644 --- a/apps/daemon/internal/dispatch/suspend.go +++ b/apps/daemon/internal/dispatch/suspend.go @@ -8,12 +8,14 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" ) +// ErrRouterQuiesced also wraps failed quiesce attempts that have fenced admission. var ErrRouterQuiesced = errors.New("dispatch: router quiesced") var ErrRouterBusy = errors.New("dispatch: router has unsettled work") // Quiesce serializes against admission, then drains every admitted output and -// receipt before acknowledging suspension. Busy rejection leaves admission open; -// a drain timeout keeps it closed until the caller shuts the connection down. +// receipt while retaining settled idle Executors for suspension. Busy rejection +// leaves admission open; a drain timeout keeps it closed until the caller shuts +// the connection down. func (r *Router) Quiesce(ctx context.Context, request proto.EnvironmentSuspendPayload) error { if strings.TrimSpace(request.EnvironmentID) == "" || strings.TrimSpace(request.SuspendID) == "" || len(request.SuspendID) > 128 { return errors.New("dispatch: invalid suspension identity") @@ -54,10 +56,12 @@ func (r *Router) Quiesce(ctx context.Context, request proto.EnvironmentSuspendPa } } for _, owner := range r.executors { - owner.idleLease++ if owner.timer != nil { owner.timer.Stop() } + // Stop cannot retract a callback already waiting for the lock. Fence its + // lease before releasing the lock, including after an eventual Resume. + owner.idleLease++ } r.mu.Unlock() err := r.shutdownWG.waitContext(ctx) @@ -65,9 +69,11 @@ func (r *Router) Quiesce(ctx context.Context, request proto.EnvironmentSuspendPa if r.closed { err = ErrRouterClosed } - r.mu.Unlock() - return err + if err != nil { + return errors.Join(ErrRouterQuiesced, err) + } + return nil } // Resume opens admission only after the caller authenticated a new connection diff --git a/apps/daemon/internal/dispatch/suspend_test.go b/apps/daemon/internal/dispatch/suspend_test.go index 054a031ed..36562c3e2 100644 --- a/apps/daemon/internal/dispatch/suspend_test.go +++ b/apps/daemon/internal/dispatch/suspend_test.go @@ -15,13 +15,22 @@ type suspendSender func(context.Context, proto.Envelope) error func (s suspendSender) Send(ctx context.Context, env proto.Envelope) error { return s(ctx, env) } -type suspendedExecutor struct{ closed atomic.Int32 } +type suspendedExecutor struct { + closed atomic.Int32 + close func(context.Context) error +} func (e *suspendedExecutor) StartTurn(context.Context, string, proto.MessageInput, chan<- proto.Envelope) (agent.Turn, error) { return nil, errors.New("suspended executor must not start a Turn") } -func (e *suspendedExecutor) Close(context.Context) error { e.closed.Add(1); return nil } +func (e *suspendedExecutor) Close(ctx context.Context) error { + e.closed.Add(1) + if e.close != nil { + return e.close(ctx) + } + return nil +} func suspensionRouter(t *testing.T, sender Sender) *Router { t.Helper() @@ -39,12 +48,19 @@ func suspensionRouter(t *testing.T, sender Sender) *Router { func TestQuiesceRejectsEveryUnsettledResource(t *testing.T) { cases := map[string]func(*Router){ - "active": func(r *Router) { r.sessions["run"] = &sessionState{} }, - "preparing": func(r *Router) { r.preparations["p"] = &preparationState{owns: true} }, - "receipt": func(r *Router) { r.preparations["p"] = &preparationState{busy: true} }, - "read": func(r *Router) { r.workspaceReads = map[string]struct{}{"read": {}} }, - "write": func(r *Router) { r.workspaceWrite = &workspaceUpload{} }, - "export": func(r *Router) { r.workspaceExport = &workspaceExport{} }, + "active": func(r *Router) { r.sessions["run"] = &sessionState{} }, + "preparing": func(r *Router) { r.preparations["p"] = &preparationState{owns: true} }, + "receipt": func(r *Router) { r.preparations["p"] = &preparationState{busy: true} }, + "read": func(r *Router) { r.workspaceReads = map[string]struct{}{"read": {}} }, + "write": func(r *Router) { r.workspaceWrite = &workspaceUpload{} }, + "export": func(r *Router) { r.workspaceExport = &workspaceExport{} }, + "closing executor": func(r *Router) { r.executors["s"] = &executorState{invalid: true, environmentID: "env"} }, + "preparing executor": func(r *Router) { r.executors["s"] = &executorState{preparing: true, environmentID: "env"} }, + "active executor": func(r *Router) { r.executors["s"] = &executorState{run: &sessionState{}, environmentID: "env"} }, + "admitted executor": func(r *Router) { + r.executors["s"] = &executorState{admission: &preparationState{}, environmentID: "env"} + }, + "foreign executor": func(r *Router) { r.executors["s"] = &executorState{environmentID: "other"} }, } for name, setup := range cases { t.Run(name, func(t *testing.T) { @@ -60,6 +76,7 @@ func TestQuiesceRejectsEveryUnsettledResource(t *testing.T) { r.workspaceReads = nil r.workspaceWrite = nil r.workspaceExport = nil + r.executors = map[string]*executorState{} }) } } @@ -106,7 +123,7 @@ func TestQuiesceDrainsPendingReceiptAndFencesConcurrentAdmission(t *testing.T) { } } -func TestQuiescePreservesIdleExecutorAgainstExpiredTimerAndRequiresExactResume(t *testing.T) { +func TestQuiesceRetainsIdleExecutorAndRequiresExactResume(t *testing.T) { sender := suspendSender(func(context.Context, proto.Envelope) error { return nil }) r := suspensionRouter(t, sender) native := &suspendedExecutor{} @@ -117,33 +134,78 @@ func TestQuiescePreservesIdleExecutorAgainstExpiredTimerAndRequiresExactResume(t oldLease := owner.idleLease r.mu.Unlock() request := proto.EnvironmentSuspendPayload{EnvironmentID: "env", SuspendID: "attempt"} - if err := r.Quiesce(context.Background(), request); err != nil { + if err := r.Quiesce(t.Context(), request); err != nil { t.Fatal(err) } r.expireIdleExecutor(owner, oldLease) if native.closed.Load() != 0 { - t.Fatal("pre-snapshot timer closed retained owner") + t.Fatal("quiesce or stale timer closed the resident owner") } - wrong := request - wrong.SuspendID = "obsolete" - if err := r.Resume(wrong, sender); err == nil { - t.Fatal("stale operation reopened admission") + for _, wrong := range []proto.EnvironmentSuspendPayload{ + {EnvironmentID: "env", SuspendID: "obsolete"}, + {EnvironmentID: "foreign", SuspendID: "attempt"}, + } { + if err := r.Resume(wrong, sender); err == nil { + t.Fatal("wrong suspension reopened admission") + } + } + if err := r.Resume(request, nil); err == nil { + t.Fatal("resume accepted no sender") } if err := r.Resume(request, sender); err != nil { t.Fatal(err) } + // A callback queued before Stop may run after Resume. It must not close + // the same owner under its renewed idle lease. + r.expireIdleExecutor(owner, oldLease) r.mu.Lock() - newLease := owner.idleLease + retained := r.executors[owner.sessionID] == owner && !owner.invalid && owner.native == native && owner.idleLease > oldLease + lease := owner.idleLease r.mu.Unlock() - r.expireIdleExecutor(owner, newLease) + if !retained || native.closed.Load() != 0 { + t.Fatal("resume replaced or retired the resident owner") + } + r.expireIdleExecutor(owner, lease) if native.closed.Load() != 1 { - t.Fatal("normal idle expiration was not restored") + t.Fatal("resumed idle expiry did not close the exact owner") + } +} + +func TestQuiesceStopsIdleTimerUntilResume(t *testing.T) { + sender := suspendSender(func(context.Context, proto.Envelope) error { return nil }) + r := suspensionRouter(t, sender) + closed := make(chan struct{}) + native := &suspendedExecutor{close: func(context.Context) error { close(closed); return nil }} + owner := &executorState{id: "executor", sessionID: "session", environmentID: "env", native: native, cancel: func() {}} + r.mu.Lock() + r.idleTimeout = 40 * time.Millisecond + r.executors[owner.sessionID] = owner + r.scheduleExecutorIdleLocked(owner) + r.mu.Unlock() + request := proto.EnvironmentSuspendPayload{EnvironmentID: "env", SuspendID: "attempt"} + if err := r.Quiesce(t.Context(), request); err != nil { + t.Fatal(err) + } + select { + case <-closed: + t.Fatal("parked owner's idle timer closed native resources") + case <-time.After(2 * r.idleTimeout): + } + if err := r.Resume(request, sender); err != nil { + t.Fatal(err) + } + select { + case <-closed: + case <-time.After(time.Second): + t.Fatal("resume did not restore idle expiry") } } func TestShutdownDestroysQuiescedOwnerAndCannotResume(t *testing.T) { sender := suspendSender(func(context.Context, proto.Envelope) error { return nil }) r := suspensionRouter(t, sender) + native := &suspendedExecutor{} + r.executors["session"] = &executorState{id: "executor", sessionID: "session", environmentID: "env", native: native, cancel: func() {}} request := proto.EnvironmentSuspendPayload{EnvironmentID: "env", SuspendID: "attempt"} if err := r.Quiesce(context.Background(), request); err != nil { t.Fatal(err) @@ -154,6 +216,9 @@ func TestShutdownDestroysQuiescedOwnerAndCannotResume(t *testing.T) { if !errors.Is(r.Resume(request, sender), ErrRouterClosed) { t.Fatal("closed Router resurrected") } + if native.closed.Load() != 1 || len(r.executors) != 0 { + t.Fatal("shutdown did not retire the retained native owner") + } } func TestQuiesceDrainDeadlineCannotReopenAdmission(t *testing.T) { @@ -174,3 +239,130 @@ func TestQuiesceDrainDeadlineCannotReopenAdmission(t *testing.T) { t.Fatal(err) } } + +func TestQuiescedShutdownRetainsFailedCloseForRetry(t *testing.T) { + failure := errors.New("native cleanup failed") + native := &suspendedExecutor{close: func(context.Context) error { return failure }} + r := suspensionRouter(t, suspendSender(func(context.Context, proto.Envelope) error { return nil })) + owner := &executorState{id: "executor", sessionID: "session", environmentID: "env", native: native, cancel: func() {}} + r.executors[owner.sessionID] = owner + if err := r.Quiesce(t.Context(), proto.EnvironmentSuspendPayload{EnvironmentID: "env", SuspendID: "attempt"}); err != nil || native.closed.Load() != 0 { + t.Fatalf("quiesce retired owner: err=%v closes=%d", err, native.closed.Load()) + } + if err := r.Shutdown(t.Context()); !errors.Is(err, failure) { + t.Fatalf("shutdown lost native close failure: %v", err) + } + r.mu.Lock() + retained := r.executors[owner.sessionID] == owner && owner.invalid && owner.closeErr == failure + r.mu.Unlock() + if !retained { + t.Fatal("failed close abandoned ownership") + } + native.close = nil + if err := r.Shutdown(t.Context()); err != nil { + t.Fatal(err) + } + if native.closed.Load() != 2 { + t.Fatal("shutdown did not retry exactly the settled failure") + } +} + +func TestQuiesceCancelledDrainRetainsOwnerUntilShutdownSettles(t *testing.T) { + entered, release := make(chan struct{}), make(chan struct{}) + native := &suspendedExecutor{close: func(context.Context) error { close(entered); <-release; return nil }} + r := suspensionRouter(t, suspendSender(func(context.Context, proto.Envelope) error { return nil })) + t.Cleanup(func() { + select { + case <-release: + default: + close(release) + } + }) + owner := &executorState{id: "executor", sessionID: "session", environmentID: "env", native: native, cancel: func() {}} + r.executors[owner.sessionID] = owner + r.shutdownWG.Add(1) + ctx, cancel := context.WithCancel(t.Context()) + cancel() + if err := r.Quiesce(ctx, proto.EnvironmentSuspendPayload{EnvironmentID: "env", SuspendID: "attempt"}); !errors.Is(err, context.Canceled) { + t.Fatalf("quiesce=%v", err) + } + if native.closed.Load() != 0 { + t.Fatal("cancelled drain closed resident owner") + } + if err := r.Handle(t.Context(), proto.Envelope{Type: proto.TypeExecutionPrepare, ID: "late"}); !errors.Is(err, ErrRouterQuiesced) { + t.Fatalf("deadline reopened admission: %v", err) + } + r.shutdownWG.Done() + stopped := make(chan error, 1) + go func() { stopped <- r.Shutdown(t.Context()) }() + <-entered + select { + case err := <-stopped: + t.Fatalf("shutdown abandoned in-flight close: %v", err) + case <-time.After(20 * time.Millisecond): + } + close(release) + if err := <-stopped; err != nil { + t.Fatal(err) + } + if native.closed.Load() != 1 { + t.Fatal("shutdown duplicated in-flight Close") + } +} + +func TestQuiesceRejectsAdmissionInProgress(t *testing.T) { + r := suspensionRouter(t, suspendSender(func(context.Context, proto.Envelope) error { return nil })) + r.admission.Lock() + err := r.Quiesce(t.Context(), proto.EnvironmentSuspendPayload{EnvironmentID: "env", SuspendID: "attempt"}) + r.admission.Unlock() + if !errors.Is(err, ErrRouterBusy) { + t.Fatalf("quiesce=%v", err) + } + r.mu.Lock() + fenced := r.suspension != nil + r.mu.Unlock() + if fenced { + t.Fatal("busy rejection changed suspension") + } +} + +func TestQuiesceFencesAlreadyRunningTimerAcrossResume(t *testing.T) { + sender := suspendSender(func(context.Context, proto.Envelope) error { return nil }) + r := suspensionRouter(t, sender) + native := &suspendedExecutor{} + owner := &executorState{id: "executor", sessionID: "session", environmentID: "env", native: native, cancel: func() {}} + entered, release, done := make(chan struct{}), make(chan struct{}), make(chan struct{}) + t.Cleanup(func() { + select { + case <-release: + default: + close(release) + } + <-done + }) + r.mu.Lock() + r.executors[owner.sessionID] = owner + r.scheduleExecutorIdleLocked(owner) + owner.timer.Stop() + lease := owner.idleLease + owner.timer = time.AfterFunc(0, func() { + close(entered) + <-release + r.expireIdleExecutor(owner, lease) + close(done) + }) + r.mu.Unlock() + <-entered + request := proto.EnvironmentSuspendPayload{EnvironmentID: "env", SuspendID: "attempt"} + if err := r.Quiesce(t.Context(), request); err != nil { + t.Fatal(err) + } + if err := r.Resume(request, sender); err != nil { + t.Fatal(err) + } + close(release) + <-done + if native.closed.Load() != 0 { + t.Fatal("callback already running before suspension retired resumed owner") + } +} diff --git a/apps/daemon/internal/localworkspace/binding.go b/apps/daemon/internal/localworkspace/binding.go index 0ead8710a..9d24e3555 100644 --- a/apps/daemon/internal/localworkspace/binding.go +++ b/apps/daemon/internal/localworkspace/binding.go @@ -2,6 +2,7 @@ package localworkspace import ( "errors" + "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/paths" "os" "strings" "sync" @@ -22,11 +23,21 @@ type Binding struct { writer *fileWriter capabilityMu sync.Mutex capabilityRoot string + stateRoot string } // NewWithCapabilityDirectory freezes paths selected by the Runtime operator. func NewWithCapabilityDirectory(environment, session, workspace, directory string) (*Binding, error) { - return newNativeBinding(environment, session, workspace, directory) + stateRoot, err := paths.StateDirectory() + if err != nil { + return nil, err + } + b, err := newNativeBinding(environment, session, workspace, directory) + if err != nil { + return nil, err + } + b.stateRoot = stateRoot + return b, nil } func Load() (*Binding, error) { diff --git a/apps/daemon/internal/localworkspace/native_files.go b/apps/daemon/internal/localworkspace/native_files.go index e669bc66a..7748ca783 100644 --- a/apps/daemon/internal/localworkspace/native_files.go +++ b/apps/daemon/internal/localworkspace/native_files.go @@ -115,12 +115,12 @@ func (b *Binding) writeNativeFile(ctx context.Context, path string, data []byte) } defer root.Close() if err = root.MkdirAll(filepath.Dir(local), 0700); err != nil { - return result, agent.ErrWorkspaceWriteRejected + return result, unpublishedWriteError(err) } temporary := ".oac-write-" + uuid.NewString() file, err := root.OpenFile(temporary, os.O_CREATE|os.O_EXCL|os.O_WRONLY, 0600) if err != nil { - return result, agent.ErrWorkspaceWriteRejected + return result, unpublishedWriteError(err) } defer root.Remove(temporary) _, err = file.Write(data) @@ -129,7 +129,7 @@ func (b *Binding) writeNativeFile(ctx context.Context, path string, data []byte) } closeErr := file.Close() if err != nil || closeErr != nil { - return result, agent.ErrWorkspaceWriteRejected + return result, unpublishedWriteError(errors.Join(err, closeErr)) } // Link publishes complete bytes without replacing an existing destination. if err = root.Link(temporary, local); err != nil { @@ -138,12 +138,23 @@ func (b *Binding) writeNativeFile(ctx context.Context, path string, data []byte) return result, agent.ErrWorkspaceWriteDirectory } return result, agent.ErrWorkspaceWriteUnsafe + } else if errors.Is(e, fs.ErrNotExist) { + return result, unpublishedWriteError(err) } return result, agent.ErrWorkspaceWriteRejected } return agent.WorkspaceWriteResult{SizeBytes: int64(len(data))}, nil } +// unpublishedWriteError applies only before this operation publishes a file. +// Parent directories or a temporary file may already have been created. +func unpublishedWriteError(err error) error { + if storageExhausted(err) { + return agent.ErrWorkspaceWriteUnavailable + } + return agent.ErrWorkspaceWriteRejected +} + const artifactFileBytes int64 = 200 << 20 const artifactBatchBytes int64 = 500 << 20 const artifactEntries = 4096 diff --git a/apps/daemon/internal/localworkspace/native_write_error_unix.go b/apps/daemon/internal/localworkspace/native_write_error_unix.go new file mode 100644 index 000000000..6469b1b69 --- /dev/null +++ b/apps/daemon/internal/localworkspace/native_write_error_unix.go @@ -0,0 +1,12 @@ +//go:build linux || darwin + +package localworkspace + +import ( + "errors" + "syscall" +) + +func storageExhausted(err error) bool { + return errors.Is(err, syscall.ENOSPC) || errors.Is(err, syscall.EDQUOT) +} diff --git a/apps/daemon/internal/localworkspace/native_write_error_unix_test.go b/apps/daemon/internal/localworkspace/native_write_error_unix_test.go new file mode 100644 index 000000000..9c4b62370 --- /dev/null +++ b/apps/daemon/internal/localworkspace/native_write_error_unix_test.go @@ -0,0 +1,28 @@ +//go:build linux || darwin + +package localworkspace + +import ( + "errors" + "fmt" + "os" + "syscall" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/agent" +) + +func TestNativeUnpublishedWriteStorageFailure(t *testing.T) { + for _, cause := range []error{syscall.ENOSPC, syscall.EDQUOT} { + for _, err := range []error{cause, &os.PathError{Op: "write", Path: "private/path", Err: cause}, fmt.Errorf("sync failed: %w", cause), errors.Join(syscall.EIO, cause)} { + if got := unpublishedWriteError(err); got != agent.ErrWorkspaceWriteUnavailable { + t.Fatalf("storage exhaustion: got %v, want unavailable", got) + } + } + } + for _, cause := range []error{syscall.EIO, syscall.EACCES, syscall.EEXIST, syscall.ENOTDIR} { + if got := unpublishedWriteError(&os.PathError{Op: "write", Path: "private/path", Err: cause}); got != agent.ErrWorkspaceWriteRejected { + t.Fatalf("changed ordinary rejection: %v", got) + } + } +} diff --git a/apps/daemon/internal/localworkspace/native_write_error_windows.go b/apps/daemon/internal/localworkspace/native_write_error_windows.go new file mode 100644 index 000000000..b577072d4 --- /dev/null +++ b/apps/daemon/internal/localworkspace/native_write_error_windows.go @@ -0,0 +1,11 @@ +package localworkspace + +import ( + "errors" + + "golang.org/x/sys/windows" +) + +func storageExhausted(err error) bool { + return errors.Is(err, windows.ERROR_DISK_FULL) || errors.Is(err, windows.ERROR_HANDLE_DISK_FULL) || errors.Is(err, windows.ERROR_DISK_QUOTA_EXCEEDED) +} diff --git a/apps/daemon/internal/localworkspace/native_write_error_windows_test.go b/apps/daemon/internal/localworkspace/native_write_error_windows_test.go new file mode 100644 index 000000000..c250a0cd3 --- /dev/null +++ b/apps/daemon/internal/localworkspace/native_write_error_windows_test.go @@ -0,0 +1,20 @@ +package localworkspace + +import ( + "os" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/agent" + "golang.org/x/sys/windows" +) + +func TestNativeUnpublishedWriteStorageFailure(t *testing.T) { + for _, cause := range []error{windows.ERROR_DISK_FULL, windows.ERROR_HANDLE_DISK_FULL, windows.ERROR_DISK_QUOTA_EXCEEDED} { + if got := unpublishedWriteError(&os.PathError{Op: "write", Path: "private/path", Err: cause}); got != agent.ErrWorkspaceWriteUnavailable { + t.Fatalf("storage exhaustion: got %v, want unavailable", got) + } + } + if got := unpublishedWriteError(windows.ERROR_ACCESS_DENIED); got != agent.ErrWorkspaceWriteRejected { + t.Fatalf("changed ordinary rejection: %v", got) + } +} diff --git a/apps/daemon/internal/localworkspace/runtime_initialization_process.go b/apps/daemon/internal/localworkspace/runtime_initialization_process.go index 68174485d..a8464ed97 100644 --- a/apps/daemon/internal/localworkspace/runtime_initialization_process.go +++ b/apps/daemon/internal/localworkspace/runtime_initialization_process.go @@ -101,15 +101,13 @@ func initializationEnvironment(configured map[string]string) []string { // The shared process owner settles the leader and descendants. Readers finish // before return; output is discarded with constant memory, never put in errors. func runInitializationProcess(ctx context.Context, binary string, args []string, directory string, env []string) error { - operation, cancel := context.WithTimeout(ctx, 30*time.Minute) - defer cancel() - if operation.Err() != nil { + if ctx.Err() != nil { return ErrInitializationUnconfirmed } if info, err := os.Stat(directory); err != nil || !info.IsDir() { return &InitializationFailure{} } - process, err := clirunner.Start(clirunner.StartOptions{Parent: operation, Binary: binary, Args: args, + process, err := clirunner.Start(clirunner.StartOptions{Parent: ctx, Binary: binary, Args: args, Dir: directory, Env: env, KillTimeout: 250 * time.Millisecond}) if err != nil { return ErrInitializationUnconfirmed @@ -127,7 +125,7 @@ func runInitializationProcess(ctx context.Context, binary string, args []string, } first, second := <-finished, <-finished _ = process.Wait() - if operation.Err() != nil || first != nil || second != nil || process.Cmd.ProcessState == nil { + if ctx.Err() != nil || first != nil || second != nil || process.Cmd.ProcessState == nil { return ErrInitializationUnconfirmed } code := process.Cmd.ProcessState.ExitCode() diff --git a/apps/daemon/internal/localworkspace/runtime_initialization_test.go b/apps/daemon/internal/localworkspace/runtime_initialization_test.go index e16e46787..47896582a 100644 --- a/apps/daemon/internal/localworkspace/runtime_initialization_test.go +++ b/apps/daemon/internal/localworkspace/runtime_initialization_test.go @@ -348,7 +348,7 @@ func TestRuntimePreparationRejectsFilesAfterFinalization(t *testing.T) { t.Fatal(err) } digest := sha256.Sum256(nil) - input := proto.RuntimePreparePayload{Step: "begin", EnvironmentID: b.environment, SessionID: b.capabilityIdentity().SessionID, + input := proto.RuntimePreparePayload{BudgetMS: 300000, Step: "begin", EnvironmentID: b.environment, SessionID: b.capabilityIdentity().SessionID, Action: "file", File: &proto.RuntimeInitialFile{Path: "/workspace/file"}, SHA256: hex.EncodeToString(digest[:])} if err = b.ApplyRuntimePreparation(t.Context(), input, nil); !errors.Is(err, agentcapabilities.ErrInvalid) { t.Fatal("finalized Runtime accepted file", err) @@ -361,3 +361,25 @@ func TestRuntimePreparationRejectsFilesAfterFinalization(t *testing.T) { t.Fatal("finalized Runtime accepted initialize", err) } } + +func TestRuntimeInitializationDeadlineStopsDescendants(t *testing.T) { + binary, err := os.Executable() + if err != nil { + t.Fatal(err) + } + directory := t.TempDir() + ctx, cancel := context.WithTimeout(t.Context(), 500*time.Millisecond) + defer cancel() + env := append(initializationEnvironment(nil), "OAC_INITIALIZATION_FIXTURE=parent", "OAC_INITIALIZATION_FIXTURE_DIR="+directory) + err = runInitializationProcess(ctx, binary, []string{"-test.run=^TestRuntimeInitializationChild$"}, directory, env) + if !errors.Is(err, ErrInitializationUnconfirmed) || !errors.Is(ctx.Err(), context.DeadlineExceeded) { + t.Fatal("deadline outcome", err, ctx.Err()) + } + if _, err := os.Stat(filepath.Join(directory, "started")); err != nil { + t.Fatal("fixture did not start", err) + } + time.Sleep(1100 * time.Millisecond) + if _, err := os.Stat(filepath.Join(directory, "late")); !os.IsNotExist(err) { + t.Fatal("descendant survived deadline") + } +} diff --git a/apps/daemon/internal/localworkspace/snapshot_marker.go b/apps/daemon/internal/localworkspace/snapshot_marker.go index fb95e9de1..48ea5e49e 100644 --- a/apps/daemon/internal/localworkspace/snapshot_marker.go +++ b/apps/daemon/internal/localworkspace/snapshot_marker.go @@ -6,7 +6,6 @@ import ( "io" "os" - "github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/paths" "github.com/MiniMax-AI/OpenAgentCore/internal/agentcapabilities" "github.com/MiniMax-AI/OpenAgentCore/internal/runtimefs" "github.com/google/uuid" @@ -30,8 +29,11 @@ func (b *Binding) openSnapshotMarker() (*snapshotMarker, error) { return nil, agentcapabilities.ErrInvalid } } - private, err := paths.Root() - if err != nil || agentcapabilities.ValidateLocalDirectories([]string{private}) != nil { + private := b.stateRoot + if agentcapabilities.ValidateLocalDirectories([]string{private}) != nil { + return nil, agentcapabilities.ErrInvalid + } + if err := os.MkdirAll(private, 0700); err != nil { return nil, agentcapabilities.ErrInvalid } current, err := os.OpenRoot(private) diff --git a/apps/daemon/internal/localworkspace/snapshot_marker_test.go b/apps/daemon/internal/localworkspace/snapshot_marker_test.go index b57484af7..90132e1e5 100644 --- a/apps/daemon/internal/localworkspace/snapshot_marker_test.go +++ b/apps/daemon/internal/localworkspace/snapshot_marker_test.go @@ -28,7 +28,7 @@ func markerBinding(t *testing.T) (*Binding, proto.PromptRequestPayload) { } func markerPath(b *Binding) string { identity := b.capabilityIdentity() - return filepath.Join(os.Getenv("OAC_RUNTIME_HOME"), "daemon", "capability-installations", identity.EnvironmentID+"-"+identity.SessionID+".json") + return filepath.Join(b.stateRoot, "daemon", "capability-installations", identity.EnvironmentID+"-"+identity.SessionID+".json") } func TestSnapshotMarkerRoundTrip(t *testing.T) { b, _ := markerBinding(t) @@ -186,7 +186,7 @@ func TestSnapshotMarkerRefusesDamagedState(t *testing.T) { func TestCapabilityFinalizeRecordsCompletionAndRejectsLaterImports(t *testing.T) { b, _ := markerBinding(t) identity := b.capabilityIdentity() - finalize := proto.RuntimePreparePayload{Step: "begin", EnvironmentID: identity.EnvironmentID, SessionID: identity.SessionID, Action: "finalize", Sources: &agentcapabilities.Input{}} + finalize := proto.RuntimePreparePayload{BudgetMS: 300000, Step: "begin", EnvironmentID: identity.EnvironmentID, SessionID: identity.SessionID, Action: "finalize", Sources: &agentcapabilities.Input{}} if err := b.ApplyRuntimePreparation(t.Context(), finalize, nil); err != nil { t.Fatal(err) } @@ -197,7 +197,7 @@ func TestCapabilityFinalizeRecordsCompletionAndRejectsLaterImports(t *testing.T) if err := os.RemoveAll(b.capabilityRoot); err != nil { t.Fatal(err) } - skill := proto.RuntimePreparePayload{Step: "begin", EnvironmentID: identity.EnvironmentID, SessionID: identity.SessionID, Action: "skill", Skill: &agentskill.Metadata{Type: "inline", Name: "example", Description: "Example"}, SizeBytes: 1, SHA256: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"} + skill := proto.RuntimePreparePayload{BudgetMS: 300000, Step: "begin", EnvironmentID: identity.EnvironmentID, SessionID: identity.SessionID, Action: "skill", Skill: &agentskill.Metadata{Type: "inline", Name: "example", Description: "Example"}, SizeBytes: 1, SHA256: "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"} if err := b.ApplyRuntimePreparation(t.Context(), skill, []byte("x")); err == nil { t.Fatal("completed identity accepted import") } @@ -220,3 +220,34 @@ func TestSnapshotMarkerCancellationDoesNotPublishCompletion(t *testing.T) { t.Fatal("cancelled preparation published marker", err) } } + +func TestCapabilityMarkerSurvivesComputeReplacement(t *testing.T) { + retained := t.TempDir() + if err := os.Chmod(retained, 0700); err != nil { + t.Fatal(err) + } + t.Setenv("OAC_RUNTIME_STATE_DIRECTORY", retained) + b, _ := markerBinding(t) + marker, err := b.openSnapshotMarker() + if err != nil { + t.Fatal(err) + } + if err = marker.complete(); err != nil { + t.Fatal(err) + } + marker.close() + t.Setenv("OAC_RUNTIME_HOME", t.TempDir()) + identity := b.capabilityIdentity() + replacement, err := NewWithCapabilityDirectory(identity.EnvironmentID, identity.SessionID, b.workspace, b.capabilityRoot) + if err != nil { + t.Fatal(err) + } + marker, err = replacement.openSnapshotMarker() + if err != nil { + t.Fatal(err) + } + defer marker.close() + if !marker.completed { + t.Fatal("compute replacement forgot completed capability installation") + } +} diff --git a/apps/daemon/internal/paths/paths.go b/apps/daemon/internal/paths/paths.go index accdd85c6..d6af78594 100644 --- a/apps/daemon/internal/paths/paths.go +++ b/apps/daemon/internal/paths/paths.go @@ -50,6 +50,18 @@ func Root() (string, error) { return filepath.Join(home, ".oac"), nil } +// StateDirectory is the single process setting for private durable Runtime state. +// An explicit invalid value is an error, never a reason to use the default. +func StateDirectory() (string, error) { + if value, present := os.LookupEnv("OAC_RUNTIME_STATE_DIRECTORY"); present { + if !filepath.IsAbs(value) || filepath.Clean(value) != value { + return "", fmt.Errorf("OAC_RUNTIME_STATE_DIRECTORY must be a clean absolute directory") + } + return value, nil + } + return Root() +} + // ProfileDir returns ~/.oac/daemon/. It is not created here. func ProfileDir(profile string) (string, error) { if err := ValidateProfile(profile); err != nil { diff --git a/apps/daemon/internal/paths/paths_test.go b/apps/daemon/internal/paths/paths_test.go index cd6fc605f..4b355a6c5 100644 --- a/apps/daemon/internal/paths/paths_test.go +++ b/apps/daemon/internal/paths/paths_test.go @@ -107,3 +107,25 @@ func TestRootRejectsRelativeOverride(t *testing.T) { t.Fatal("relative private home accepted") } } + +func TestStateDirectorySingleSetting(t *testing.T) { + home := withTempHome(t) + t.Setenv("OAC_RUNTIME_STATE_DIRECTORY", "") + if err := os.Unsetenv("OAC_RUNTIME_STATE_DIRECTORY"); err != nil { + t.Fatal(err) + } + if got, err := paths.StateDirectory(); err != nil || got != home { + t.Fatal("default state directory", got, err) + } + selected := t.TempDir() + t.Setenv("OAC_RUNTIME_STATE_DIRECTORY", selected) + if got, err := paths.StateDirectory(); err != nil || got != selected { + t.Fatal("explicit state directory", got, err) + } + for _, invalid := range []string{"", "relative", selected + string(filepath.Separator) + ".."} { + t.Setenv("OAC_RUNTIME_STATE_DIRECTORY", invalid) + if _, err := paths.StateDirectory(); err == nil { + t.Fatal("invalid explicit setting fell back to default") + } + } +} diff --git a/contracts/agents-api/environment-files.md b/contracts/agents-api/environment-files.md index 783da51fe..4f00b04f9 100644 --- a/contracts/agents-api/environment-files.md +++ b/contracts/agents-api/environment-files.md @@ -77,7 +77,7 @@ Unless the table names a code, the 400 errors have type and code `invalid_reques - Before sending any bytes, Core records the write under the Session lock. While input is pending, a Turn is running or an earlier write is unsettled, a new write returns 409 `turn_conflict`. An unsettled write also makes new messages to the Session return 409. - The Runtime checks the complete body against its digest before creating anything, so incomplete input creates nothing. A write that fails later can leave newly created empty parent directories. -- Only a definite receipt from the Runtime settles a write, as committed or rejected. A rejected write changes nothing and releases the Session. If the connection drops, the request times out or no receipt arrives, the request returns 503 and the write stays unsettled, across Core restarts. Core never resends it and has no automatic recovery, so the Session accepts no further writes or messages. Reads still work. +- Only a definite receipt from the Runtime settles a write, as committed or rejected. A rejected write publishes no destination file and releases the Session; newly created empty parent directories can remain. Storage or quota exhaustion confirmed before publication returns 503, while malformed input and destination conflicts retain their existing errors. If the connection drops, the request times out or no receipt arrives, the request returns 503 and the write stays unsettled, across Core restarts. Core never resends it and has no automatic recovery, so the Session accepts no further writes or messages. Reads still work. - Deleting the source File after its bytes were read does not affect the copy. ## Artifacts diff --git a/contracts/agents-api/environments.md b/contracts/agents-api/environments.md index 4ac2eece0..79126cbc7 100644 --- a/contracts/agents-api/environments.md +++ b/contracts/agents-api/environments.md @@ -92,6 +92,8 @@ The application owns the machine. It creates the Session with a clean absolute ` Session input goes through `POST /agents/sessions/{id}/events` in ordered batches of 1–64 events. On an Environment-bearing Session, Core reserves input that cannot start yet until the Environment is connected and prepared. +A valid node-backed creation can wait for compute capacity after committing. Initial input uses the same reservation and deadline while waiting; creating without input still requests Environment preparation. [Placement and generation selection](./sandbox-deployment.md#generation-ownership-and-rollout) define the scheduling rules. + ### Initial input Creation accepts initial text as a string or an ordered array of user messages. Omitted or null input creates no Turn and no connection action. @@ -211,10 +213,12 @@ The daemon runs as its launching account and never uses sudo or raises its permi - Setup commands run with Bash; on Windows, Git Bash is required and no other shell substitutes. The default working directory is `/workspace`. - On Windows, npm installation and stdio MCP commands named `npm` or `npx` (including their `.cmd` shims) run through npm's JavaScript entry point with Node, without an extra shell. -Initialization and package directories default to `initialization` and `packages` under the Runtime home (`OAC_RUNTIME_HOME`) and can be set with `OAC_RUNTIME_INITIALIZATION_DIRECTORY` and `OAC_RUNTIME_PACKAGE_DIRECTORY`; packaged Linux images use `/environment/initialization` and `/environment/packages`. These are resource paths, never Environment-source or operating-system switches in Core. +[Runtime resource directories](../../docs/configuration.md#runtime-resource-directories) define the initialization, package and private state locations. These are resource paths, never Environment-source or operating-system switches in Core. Every command uses the launching user's permissions and the host network. Process ownership waits for exit and I/O settlement. Command output is discarded; a confirmed failure keeps only a bounded integer exit status. +When hosted compute uses external workspace storage and is replaced, only the retained Environment filesystem and the qualified Harness's private native history persist. Setup commands are not replayed. Setup and tools can write other locations permitted to the launching user, but changes on the temporary system disk, process memory and background processes are not preserved by cold continuation. Store durable outputs in the workspace and use the declared package installation paths; the [Sandbox Provider lifecycle](../../docs/sandbox-provider.md#suspension) defines when compute can be replaced. + ### Explicit local tool environment The installer's `--tool-env-file` (`OAC_RUNTIME_TOOL_ENV_FILE`) supplies the Runtime operator's base tool variables. Preparation copies these values into its private initialization snapshot, and the Session's `env` keys override them. The Runtime never rewrites the source file or inherits unrelated ambient credentials. Setup, capability resolution and Harness execution read the same prepared snapshot. Reconnecting keeps that snapshot even if the operator edits the file; a new Session reads the current file. A Harness profile may reference the Runtime-owned file but must not persist copies of its values. A missing or invalid configured file fails preparation. @@ -225,7 +229,7 @@ An Environment's initialization is `pending`, `running`, `complete` or `failed`, - The Worker's initialization scheduler scans 32 Environments at a time, wraps at the end and bounds concurrent preparations by execution concurrency, independently of Provider maintenance. - A missing socket does not consume a pending attempt. An unavailable Harness fails before installation. Each operation rechecks current authority and the original socket; completion rechecks the exact binding. -- Each file transfer, configure, Skill, Plugin, package and setup step has a two-minute budget; the whole initialization has 30 minutes. Initial input keeps its five-minute admission deadline, so large installations should start from an idle Session. +- Each file, Skill or Plugin transfer has a two-minute staging budget. Configure, package installation, setup commands and snapshot finalization share the whole initialization’s 30-minute budget, whose remaining duration Core carries on each operation and the Runtime enforces. Initial input keeps its five-minute admission deadline, so large installations should start from an idle Session. - A running initialization whose owner is lost, including across a Core restart, fails as unconfirmed; nothing is replayed. A completed Environment never reinstalls on reconnect or native recovery, so later user changes survive. - Failure is terminal for the Session but destroys neither compute nor files. diff --git a/contracts/agents-api/harness-capabilities.md b/contracts/agents-api/harness-capabilities.md index c1ad61d8f..b5130ae51 100644 --- a/contracts/agents-api/harness-capabilities.md +++ b/contracts/agents-api/harness-capabilities.md @@ -59,3 +59,5 @@ These operations need a workspace, so they apply to hosted and self-hosted place | Network `disabled` or `restricted` | Rejected | Rejected | Rejected | | [Plugin MCP over stdio](./environments.md#plugin-mcp) | Verified: hosted, self-hosted | Verified: self-hosted; admitted: hosted | Verified: self-hosted; admitted: hosted | | Plugin MCP over HTTP | Admitted, with literal headers or HTTPS bearer | Admitted, anonymous or HTTPS bearer | Verified: self-hosted; admitted: hosted; anonymous or HTTPS bearer | + +External workspace storage additionally requires the selected Harness profile and Runtime to support `retained_native_history`; owned workspace storage does not. The [Workspace Provider guide](../../docs/workspace-provider.md) owns storage combination admission. diff --git a/contracts/agents-api/harness-onboarding.md b/contracts/agents-api/harness-onboarding.md index 2bae3fb5f..d473444ef 100644 --- a/contracts/agents-api/harness-onboarding.md +++ b/contracts/agents-api/harness-onboarding.md @@ -102,7 +102,7 @@ The service profile qualifies public combinations and the Runtime advertises the A Session owns one reusable Executor in its connected Runtime; a Turn owns one input execution, its output stream and its cancellation. `agent.ExecutorFactory` prepares the fixed configuration without model input, and `Executor.StartTurn` creates a new `agent.Turn` without replacing healthy native resources. Normal completion settles only the Turn. `Executor.Close` releases native resources on idle expiry, Environment shutdown or confirmed invalidation; it releases neither the Environment allocation nor the workspace. Core keeps no second Executor cache. The same lifecycle applies to hosted, self-hosted and `none` placements. -**Binding.** The Runtime binds its Executor record to the Session, Environment, connection and immutable execution configuration. Resume identity and prior-Turn recovery flags are continuity assertions, not configuration changes. A supplied native identity must match the retained owner, and when existing history is required, recovery never starts a new root. A configuration conflict is an error, not a hot switch. A lost connection retires its owners and handles; old timers, output and cancellation cannot affect their replacements. +**Binding.** The Runtime binds its Executor record to the Session, Environment, connection and immutable execution configuration. Resume identity and prior-Turn recovery flags are continuity assertions, not configuration changes. A supplied native identity must match the retained owner, and when existing history is required, recovery never starts a new root. A configuration conflict is an error, not a hot switch. An ordinary lost connection retires its owners and handles; [planned suspension](../../docs/runtime-protocol.md#preparation-and-execution-order) retains settled idle owners through exact authenticated resume. Old timers, output and cancellation cannot affect replacement owners. **Per-Turn state.** Each Turn gets a fresh wrapper, output channel and receipt state. Steering and function interfaces belong to that Turn. Native callbacks capture the originating Turn before asynchronous work, so a late event is never attributed to whichever Turn is active. Native processes, query or transport connections, fixed capability configuration and native session identity belong to the Executor. Do not reset completed `sync.Once` values or reuse an old Turn object. @@ -147,7 +147,9 @@ A Harness that supports the Subagent reads implements the [neutral observation c ## Register the adapter -Registration is static and requires a build. Export one `agent.Declaration` from `apps/daemon/internal/agent//declaration.go`, then add it to `harnessDeclarations` in [`cli/agent_discovery.go`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/apps/daemon/internal/cli/agent_discovery.go). The declaration contains the kind and complete capability descriptor, the shared model `Configuration` and a `Discover` function. Discovery receives the profile and diagnostic writers, owns native configuration and availability checks, and returns the installed `agent.Runtime` with its descriptor and Executor factory. Return nil when the adapter is not configured; return an unavailable descriptor without a factory when configured prerequisites fail. Keep version gates and factory-selection conditions inside the adapter. +Registration is static and requires a build. Export one `agent.Declaration` from `apps/daemon/internal/agent//declaration.go`, then add it to `harnessDeclarations` in [`cli/agent_discovery.go`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/apps/daemon/internal/cli/agent_discovery.go). The declaration contains the kind and complete capability descriptor, the shared model `Configuration` and a `Discover` function. Discovery receives the profile, diagnostic writers and the Runtime-resolved `DiscoveryOptions.StateRoot` and instance-private `DiscoveryOptions.RuntimeRoot`, owns native configuration and availability checks, and returns the installed `agent.Runtime` with its descriptor and Executor factory. Return nil when the adapter is not configured; return an unavailable descriptor without a factory when configured prerequisites fail. Keep version gates and factory-selection conditions inside the adapter. + +The Runtime resolves the [state directory](../../docs/configuration.md#runtime-resource-directories) once and passes it explicitly to every adapter. Adapters that support retained native history place it beneath that root with their own namespaced layout; they do not choose a fallback root from ambient environment or put history in the workspace. An adapter without this capability keeps instance-private native state and must not declare history portable merely because a state root is supplied. Private launch homes, scratch data and instance-private native state use the supplied `DiscoveryOptions.RuntimeRoot`. Device credentials and connection identity stay outside retained state. A state path alone does not qualify native continuation after compute replacement: the adapter must preserve native ownership, reject missing or foreign history before model input, and complete native shutdown before another writer attaches. [`cli/agent_registration.go`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/apps/daemon/internal/cli/agent_registration.go) iterates the discovered runtimes and calls `Registry.Register` from `agent/harness.go`. It verifies that discovery retained the declared kind and registers the Runtime in this order: diff --git a/contracts/agents-api/index.md b/contracts/agents-api/index.md index 9e9f6bb7f..f5eed41ba 100644 --- a/contracts/agents-api/index.md +++ b/contracts/agents-api/index.md @@ -127,6 +127,7 @@ Each item is Core's deliberate or native behavior where the official service beh - Runtimes do not enforce `disabled` or `restricted` networks, so Sessions that need them are rejected ([restricted network policy](./environments.md#restricted-network)). - `packages.system` is rejected; system packages must be preinstalled. +- Cold continuation does not restore process memory, background processes or temporary system-disk changes; completed setup is not replayed. See [Environment persistence](./environments.md#runtime-capability-preparation). Cross-node memory restoration and takeover of an unconfirmed old writer are not qualified. **Files and Environment files** diff --git a/contracts/agents-api/node-generation-protocol.md b/contracts/agents-api/node-generation-protocol.md index cc2d0f76f..74e338367 100644 --- a/contracts/agents-api/node-generation-protocol.md +++ b/contracts/agents-api/node-generation-protocol.md @@ -74,9 +74,9 @@ Node startup and generation loading validate complete Provider operation declara ## Generation control -A node without generation management serves only its enrolled generation, with fixed configuration, and receives no preparation or retention frames. A generation-managing node prepares the target generation Core announces in `welcome` and `heartbeat_ack` and keeps serving its durable serving generation while it does; target preparation is independent of the serving provider's readiness. +A node without generation management serves only its enrolled generation, with fixed configuration, and receives no preparation or retention frames. It can report that exact generation in `Health.Generations`; a checkpoint-capable provider must do so, including its verified checkpoint qualification. `provider_ready` and the reported state must agree. A generation-managing node prepares the target generation Core announces in `welcome` and `heartbeat_ack` and keeps serving its durable serving generation while it does; target preparation is independent of the serving provider's readiness. -Its `hello` and heartbeats carry at most eight generation observations. Each names a positive signed-64-bit generation, its lowercase SHA-256 specification digest, a `ready`, `preparing` or `failed` state and an optional fixed diagnostic. The target and serving generations come first; other records rotate fairly. Eight bounds one message, not the number of generations a node may keep. An omitted observation never authorizes deletion or implies absence. +Its `hello` and heartbeats carry at most eight generation observations. Each names a positive signed-64-bit generation, its lowercase SHA-256 specification digest, a `ready`, `preparing` or `failed` state and an optional fixed diagnostic. The target and serving generations come first; other records rotate fairly. Eight bounds one message, not the number of generations a node may keep. An omitted observation never authorizes deletion or implies absence. A ready checkpoint-capable generation includes `checkpoint` with nonempty opaque `artifact_domain` and `execution_class` tokens, each at most 256 bytes. Unsupported providers omit it; preparing or failed generations never advertise it. Core persists the declaration with the authenticated connection and generation, and uses it only while that generation is ready on the current online connection. The [Provider contract](../../docs/sandbox-provider.md#checkpoint-transfer) defines archive and execution compatibility. Retention uses its own bounded exchange. A `retention` request names at most eight local `(generation, specification_digest)` references, a UUID, a sequence that increases by one, the current `connection_id` and the owner epoch. The `retention_ack` must match the complete pending request, entry order and identity included, and give an explicit boolean `keep` for every entry. One exchange is pending per connection, and a disconnect discards it. An unsolicited, replayed, stale, partial or mixed acknowledgement deletes nothing. Retention traffic never uses the Provider request queue. @@ -124,6 +124,6 @@ The readiness classes, one exported error and one code each, are authored in `se Preparation diagnostics keep fixed typed causes. Only artifact transfer, checksum or release-provenance failures report `runtime_download_failed`; the private preparer signals that class through its exit category, without Core or the node parsing stderr. Provider, ownership, cancellation and unclassified failures keep their typed code or `provider_unavailable`. No raw provider text crosses the protocol. -Creation carries the Session-selected `Bootstrap.Harness` into Runtime bootstrap version 2. Core and nodes use protocol version 7 and require a coordinated upgrade. `Bootstrap.Harness` is required. All ten suspension operations belong to the same `SandboxProvider` and are declared supported or unsupported together; dispatch uses no optional interface. `Observe` reads one allocation per request. +Creation carries the Session-selected `Bootstrap.Harness` into Runtime bootstrap version 2. Core and nodes use protocol version 8 and require a coordinated upgrade. `Bootstrap.Harness` is required. All ten suspension operations belong to the same `SandboxProvider` and are declared supported or unsupported together; dispatch uses no optional interface. `Observe` reads one allocation per request. -The current wire version is 7. Create bootstrap and Resume requests may carry an optional workspace filesystem binding. The node validates its tenant and Environment against the allocation reference, its immutable ObjectID, and its attachment configuration ID against the supplied configuration before forwarding it. The filesystem resolver validates adapter-native ownership. A missing binding selects owned storage; a present binding cannot fall back. Error responses may retain a Resume target only when its nonempty native ID, name, generation and retained-state provenance match the requested target; this is cleanup evidence, never successful restore. +The current wire version is 8. Create bootstrap and Resume requests may carry an optional workspace filesystem binding. The node validates its tenant and Environment against the allocation reference, its immutable ObjectID, and its attachment configuration ID against the supplied configuration before forwarding it. The filesystem resolver validates adapter-native ownership. A missing binding selects owned storage; a present binding cannot fall back. Error responses may retain a Resume target only when its nonempty native ID, name, generation and retained-state provenance match the requested target; this is cleanup evidence, never successful restore. diff --git a/contracts/agents-api/sandbox-deployment.md b/contracts/agents-api/sandbox-deployment.md index 6105503f4..dee249586 100644 --- a/contracts/agents-api/sandbox-deployment.md +++ b/contracts/agents-api/sandbox-deployment.md @@ -122,7 +122,7 @@ No provider text or credential is returned. The write and its `change` or `repla ## Generation ownership and rollout -`runtime_deployment` holds the current specification. Superseded rows keep only immutable specification, build and endpoint metadata, never another E2B key. An E2B allocation binds its generation at reservation; a node placement binds at Session admission, and its allocation copies that generation, even across later updates. Inspection, renewal, commands and cleanup use the allocation's original specification and endpoint with the current key; a missing generation never falls back to the current specification. Released generation identifiers stay reserved. +`runtime_deployment` holds the current specification. Superseded rows keep only immutable specification, build and endpoint metadata, never another E2B key. An E2B allocation binds its generation at reservation; a node placement binds when the scheduler reserves compatible capacity, and its allocation copies that generation, even across later updates. Inspection, renewal, commands and cleanup use the allocation's original specification and endpoint with the current key; a missing generation never falls back to the current specification. Released generation identifiers stay reserved. A generation is retained while it is current, referenced by an unreleased allocation or placement, or pinned by a node that is not removed. A node's durable serving pin survives offline periods and zero resources and is separate from its current readiness. Collection shares the deployment lock with updates and admission and deletes at most 32 eligible generation rows per pass. Reset retires pins and clears superseded rows only after confirmed release. @@ -142,7 +142,7 @@ A node is online while it is connected under the current owner epoch with a hear Each node adds `rollout: {state, ready_generation, diagnostic?}`, where `ready_generation` is the nullable durable serving pin and `diagnostic` a fixed code for the target generation; allocation items add `deployment_generation`. Poll every five seconds only while `rollout.state` is `preparing` or `reset` is not null; old Sessions and failed, update-required or offline nodes alone do not keep polling active. -New admission filters nodes by online presence, exact serving-generation readiness, address and shared capacity before it prefers the newest qualifying pin, so a full newest node never hides a free older one. Without a candidate, admission creates no provisional Session or placement: an online node that is actually preparing with free capacity gives 503 `sandbox_nodes_preparing`, and a full or offline fleet gives `runtime_node_unavailable`. +Node-backed creation validates the target deployment and commits the Session, pending Environment and any initial input without reserving compute. A full, offline or preparing fleet leaves that accepted work waiting; an unconfigured deployment, reset or unsupported combination still rejects admission. The common scheduler uses bounded rotating scans of unplaced demand, with pages ordered by recorded time and Environment ID and a fixed time boundary for each scan so continuous arrivals cannot prevent it from revisiting older work. It checks online presence, exact serving-generation readiness, address, shared capacity and the selected generation's Harness/filesystem compatibility, then prefers the newest qualifying pin. Placement is immutable once reserved until confirmed release or a settled checkpoint transfer. A bounded scan skips temporarily unavailable demand and resumes after a restart. Existing suspended allocations reserve eligible checkpoint capacity through the same lock under the [checkpoint transfer contract](../../docs/sandbox-provider.md#checkpoint-transfer); this is not a global fairness guarantee across hot restores and unplaced work. Input retains its [original five-minute deadline](./environments.md#reservations), including time waiting for capacity. ## Reset diff --git a/contracts/agents-api/zh/environment-files.md b/contracts/agents-api/zh/environment-files.md index 1977f2b63..8f31c84e2 100644 --- a/contracts/agents-api/zh/environment-files.md +++ b/contracts/agents-api/zh/environment-files.md @@ -1,7 +1,7 @@ --- title: "Environment 文件与 Artifact" source: contracts/agents-api/environment-files.md -source_hash: 1b58aa02aaccddb9675ef41ebfe2506a6fba0bb12139efb67e0da378d879aee7 +source_hash: f87f06138789666b91140c15ffd104cffba4566580921b94b094bf31591d201d --- Session 工作区保存由 agent 及其工具修改的实时文件。`/agents/environments/{environment_id}/files` 列出一个工作区目录,并在其中创建文件。Turn 完成时,Core 将工作区 `outputs/` 目录中的文件复制为不可变 Artifact,通过 `/agents/sessions/{session_id}/artifacts` 读取。Artifact 的生命周期长于 Environment;工作区文件则不是。 @@ -79,7 +79,7 @@ Session 工作区保存由 agent 及其工具修改的实时文件。`/agents/en - 发送任何字节之前,Core 在 Session 锁下记录写入。输入待处理、Turn 运行或较早写入未结算时,新写入返回 409 `turn_conflict`。未结算写入也使 Session 新消息返回 409。 - Runtime 在创建任何内容前根据摘要检查完整正文,因此不完整输入不创建内容。后续失败的写入可能留下新建空父目录。 -- 仅 Runtime 的确定回执将写入结算为已提交或已拒绝。被拒绝写入不改变内容并释放 Session。连接断开、请求超时或无回执时,返回 503,写入持续未结算,Core 重启后仍如此。Core 不重发,也无自动恢复,因此 Session 不再接受写入或消息。读取仍可用。 +- 仅 Runtime 的确定回执将写入结算为已提交或已拒绝。被拒绝写入不会发布目标文件,并释放 Session;新建的空父目录可能保留。确认在发布前发生的存储或配额耗尽返回 503,格式错误的输入和目标冲突仍保持原有错误。连接断开、请求超时或无回执时,返回 503,写入持续未结算,Core 重启后仍如此。Core 不重发,也无自动恢复,因此 Session 不再接受写入或消息。读取仍可用。 - 源 File 字节读取后,删除该 File 不影响副本。 ## Artifact {#artifacts} diff --git a/contracts/agents-api/zh/environments.md b/contracts/agents-api/zh/environments.md index 1891bd481..ffed2f938 100644 --- a/contracts/agents-api/zh/environments.md +++ b/contracts/agents-api/zh/environments.md @@ -1,7 +1,7 @@ --- title: "环境与模板" source: contracts/agents-api/environments.md -source_hash: 8fb6cbd0b8daed4686d1b6829a49e799f03bb78c6bdb1f93cdb9a4518fec4f8f +source_hash: 9007be1099cf50f5ba376d2de4a060d039b6ee0b1c13c78a94665c28c2fc3fc6 --- Environment 是 Session 的执行资源,包括 Harness 运行所在的机器、工作区以及已完成准备的能力。Session 通过其 `environment` 配置创建 Environment;不存在独立的 create 调用。Environment Template 是 Session 创建时解析的可复用准备配置。本契约涵盖这两类资源、两种放置方式、输入接纳、能力准备、Skills、Plugins 和 MCP 连接来源。 @@ -94,6 +94,8 @@ Core 在 Session 创建事务中创建 Environment 记录;Session upsert 会 Session 输入通过 `POST /agents/sessions/{id}/events` 以每次 1–64 个事件的有序批次提交。对于带 Environment 的 Session,Core 会将尚不能启动的输入预留,直到 Environment 完成连接和准备。 +有效的节点模式创建可以在提交后等待计算容量。初始输入在等待期间仍使用同一 reservation 和期限;无输入创建也会请求 Environment 准备。[Placement 与代次选择](./sandbox-deployment.md#generation-ownership-and-rollout) 定义调度规则。 + ### 初始输入 {#initial-input} 创建操作接受字符串形式的初始文本,或由用户消息构成的有序数组。省略或传入 null 输入不会创建 Turn 或连接操作。 @@ -213,10 +215,12 @@ daemon 以启动它的账户身份运行,绝不使用 sudo 或提升权限。 - 设置命令使用 Bash 运行;在 Windows 上必须使用 Git Bash,且不能由其他 shell 替代。默认工作目录为 `/workspace`。 - 在 Windows 上,npm 安装以及名为 `npm` 或 `npx` 的 stdio MCP 命令(包括其 `.cmd` shim)会通过 Node 调用 npm 的 JavaScript 入口点运行,而不经过额外的 shell。 -初始化目录和软件包目录默认分别是 Runtime 主目录(`OAC_RUNTIME_HOME`)下的 `initialization` 和 `packages`,也可通过 `OAC_RUNTIME_INITIALIZATION_DIRECTORY` 和 `OAC_RUNTIME_PACKAGE_DIRECTORY` 设置;打包的 Linux 镜像使用 `/environment/initialization` 和 `/environment/packages`。这些是资源路径,在 Core 中绝不是 Environment 源或操作系统开关。 +[Runtime 资源目录](../../../docs/zh/configuration.md#runtime-resource-directories)定义初始化、软件包和私有状态的位置。这些是资源路径,在 Core 中绝不是 Environment 源或操作系统开关。 每条命令都使用启动用户的权限和主机网络。进程所有权会等待退出及 I/O 结算完成。命令输出会被丢弃;确认失败时只保留一个有界整数退出状态。 +使用外部工作区存储的托管计算替换后,仅持久 Environment 文件系统和已通过资格验证的 Harness 私有原生历史会保留。Setup 命令不会重放。Setup 和工具可以写入启动用户有权访问的其他位置,但冷续接不保留临时系统盘上的修改、进程内存或后台进程。持久输出应写入工作区,软件包应使用已声明的安装路径;[Sandbox Provider 生命周期](../../../docs/zh/sandbox-provider.md#suspension)定义何时可以替换计算。 + ### 显式本地工具环境 {#explicit-local-tool-environment} 安装器的 `--tool-env-file`(`OAC_RUNTIME_TOOL_ENV_FILE`)提供 Runtime 操作者的基础工具变量。准备过程会将这些值复制到其私有初始化快照中,并由 Session 的 `env` 键覆盖。Runtime 绝不会重写源文件,也不会继承无关的环境凭据。设置、能力解析和 Harness 执行都会读取同一份已准备快照。即使操作者编辑了文件,重连仍会保留该快照;新 Session 会读取当前文件。Harness profile 可以引用由 Runtime 所有的文件,但不得持久保存其值的副本。配置的文件缺失或无效时,准备过程会失败。 @@ -227,7 +231,7 @@ Environment 的初始化状态为 `pending`、`running`、`complete` 或 `failed - Worker 的初始化调度器每次扫描 32 个 Environment,在末尾循环回绕,并依据执行并发度限制并发准备,且独立于 Provider 维护。 - 缺少套接字不会消耗一次 pending 尝试。Harness 不可用时,会在安装前失败。每个操作都会重新检查当前权限和原始套接字;完成时还会重新检查精确绑定。 -- 每个文件传输、configure、Skill、Plugin、软件包和设置步骤都有两分钟的预算;整个初始化过程有 30 分钟。初始输入仍保留其五分钟接纳期限,因此大型安装应从空闲 Session 开始。 +- 每个文件、Skill 或 Plugin 传输的暂存预算为两分钟。configure、软件包安装、设置命令和快照最终确定共享整个初始化过程的 30 分钟预算,Core 在每项操作中携带剩余时长,由 Runtime 执行。初始输入仍保留其五分钟接纳期限,因此大型安装应从空闲 Session 开始。 - 正在运行且所有权丧失的初始化,包括跨 Core 重启丧失所有权,会作为未确认而失败;不会重播任何内容。已完成的 Environment 在重连或原生恢复时绝不会重新安装,因此用户后续更改会保留下来。 - 失败对 Session 而言是终止状态,但不会销毁计算资源或文件。 diff --git a/contracts/agents-api/zh/harness-capabilities.md b/contracts/agents-api/zh/harness-capabilities.md index b947daaf8..e95d9114d 100644 --- a/contracts/agents-api/zh/harness-capabilities.md +++ b/contracts/agents-api/zh/harness-capabilities.md @@ -1,7 +1,7 @@ --- title: "Harness 能力" source: contracts/agents-api/harness-capabilities.md -source_hash: e1ffc7a260ceaec64ba377f7c0db28f2c371c9d664098b110b740cc110506506 +source_hash: a88b7238811707dd7fecae6ee8e509572677b49958e36ddabf8b783e97382c13 --- 本页列出每个 Harness 在每种部署位置支持的能力。Core 根据 `services/core/internal/engine` 中 Harness 的引擎配置决定准入,运行 Session 的 Runtime 也必须声明操作。所链接契约定义各操作;[Harness 接入](harness-onboarding.md#qualify-the-adapter)说明资格验证方法。 @@ -61,3 +61,5 @@ Claude 结构化输出要求单 Agent、medium verbosity,且无 Skill、Plugin | 网络 `disabled` 或 `restricted` | 已拒绝 | 已拒绝 | 已拒绝 | | [stdio Plugin MCP](environments.md#plugin-mcp) | 已验证:托管、自托管 | 已验证:自托管;已准入:托管 | 已验证:自托管;已准入:托管 | | HTTP Plugin MCP | 已准入,字面值头或 HTTPS bearer | 已准入,匿名或 HTTPS bearer | 已验证:自托管;已准入:托管;匿名或 HTTPS bearer | + +外部工作区存储还要求所选 Harness profile 和 Runtime 支持 `retained_native_history`;自带工作区存储不要求此能力。[Workspace Provider 指南](../../../docs/zh/workspace-provider.md)定义存储组合准入规则。 diff --git a/contracts/agents-api/zh/harness-onboarding.md b/contracts/agents-api/zh/harness-onboarding.md index 6b7204bfb..d422a7116 100644 --- a/contracts/agents-api/zh/harness-onboarding.md +++ b/contracts/agents-api/zh/harness-onboarding.md @@ -1,7 +1,7 @@ --- title: "添加 Harness" source: contracts/agents-api/harness-onboarding.md -source_hash: 38955b517dae94e5ce187e72a6b2e03cc7eeb7b886baa722b5f117dd3260253f +source_hash: 2fe919ec4fd2285cc9e54fed58ef66aebe66d132fd8fb5adc272e26b2e673774 --- **Harness** 是一种运行模型和工具循环的原生代理引擎(Codex、Claude Code、MiniMax Code)。**Harness 适配器**将 Runtime 的 Executor 和 Turn 契约转换到该引擎的 SDK 或协议。本文档定义 Runtime–Harness 协议:适配器接口及其生命周期义务、注册、Core 资格认定和验收。[Harness capabilities](harness-capabilities.md) 记录了当前每个 Harness 支持的功能。 @@ -104,7 +104,7 @@ func (s *Session) SubmitFunctionResult(context.Context, proto.FunctionResultPayl Session 在其已连接的 Runtime 中拥有一个可复用的 Executor;Turn 拥有一次输入执行、其输出流和其取消操作。`agent.ExecutorFactory` 在没有模型输入的情况下准备固定配置,而 `Executor.StartTurn` 创建新的 `agent.Turn`,不替换健康的原生资源。正常完成仅结算 Turn。`Executor.Close` 在空闲过期、Environment 关闭或确认失效时释放原生资源;它既不释放 Environment 分配,也不释放工作区。Core 不保留第二套 Executor 缓存。相同的生命周期适用于托管、自托管和 `none` 放置方式。 -**绑定。** Runtime 将其 Executor 记录绑定到 Session、Environment、连接和不可变执行配置。恢复身份和先前 Turn 恢复标志是连续性断言,而不是配置更改。提供的原生身份必须与保留的所有者匹配;当需要现有历史时,恢复绝不能启动新的根。配置冲突属于错误,而不是热切换。连接丢失会让其所有者和句柄退役;旧计时器、输出和取消操作不能影响替代对象。 +**绑定。** Runtime 将其 Executor 记录绑定到 Session、Environment、连接和不可变执行配置。恢复身份和先前 Turn 恢复标志是连续性断言,而不是配置更改。提供的原生身份必须与保留的所有者匹配;当需要现有历史时,恢复绝不能启动新的根。配置冲突属于错误,而不是热切换。普通连接丢失会让其所有者和句柄退役;[计划性挂起](../../../docs/zh/runtime-protocol.md#preparation-and-execution-order) 则保留已结算的空闲所有者,直至精确匹配且已认证的恢复完成。旧计时器、输出和取消操作不能影响替代所有者。 **每 Turn 状态。** 每个 Turn 都会获得全新的包装器、输出通道和回执状态。引导和函数接口均属于该 Turn。原生回调必须在异步工作开始前捕获来源 Turn,因此迟到事件绝不会被归到当前活动的 Turn 上。原生进程、query 或传输连接、固定能力配置和原生 session 身份均属于 Executor。不要重置已完成的 `sync.Once` 值,也不要复用旧 Turn 对象。 @@ -149,7 +149,9 @@ MCP、公共函数、延迟函数发现、结构化输出、图像输入、详 ## 注册适配器 {#register-the-adapter} -注册是静态的,并且需要构建。从 `apps/daemon/internal/agent//declaration.go` 导出一个 `agent.Declaration`,然后将其添加到 [`cli/agent_discovery.go`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/apps/daemon/internal/cli/agent_discovery.go) 的 `harnessDeclarations` 中。声明包含 kind、完整能力描述符、共享模型 `Configuration` 和 `Discover` 函数。发现过程接收 profile 和诊断写入器,负责原生配置和可用性检查,并返回已安装的 `agent.Runtime` 及其描述符和 Executor 工厂。未配置适配器时返回 nil;已配置的前置条件失败时,返回不带工厂的不可用描述符。将版本门控和工厂选择条件保留在适配器内部。 +注册是静态的,并且需要构建。从 `apps/daemon/internal/agent//declaration.go` 导出一个 `agent.Declaration`,然后将其添加到 [`cli/agent_discovery.go`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/apps/daemon/internal/cli/agent_discovery.go) 的 `harnessDeclarations` 中。声明包含 kind、完整能力描述符、共享模型 `Configuration` 和 `Discover` 函数。发现过程接收 profile、诊断写入器以及 Runtime 解析后的 `DiscoveryOptions.StateRoot` 和实例私有的 `DiscoveryOptions.RuntimeRoot`,负责原生配置和可用性检查,并返回已安装的 `agent.Runtime` 及其描述符和 Executor 工厂。未配置适配器时返回 nil;已配置的前置条件失败时,返回不带工厂的不可用描述符。将版本门控和工厂选择条件保留在适配器内部。 + +Runtime 统一解析[状态目录](../../../docs/zh/configuration.md#runtime-resource-directories),并显式传给每个适配器。支持保留原生历史的适配器在该根目录下按各自的命名空间布局存放历史,不从环境变量选择回退根目录,也不把历史放进工作区。不支持该能力的适配器将原生状态保持为实例私有,不能仅因收到状态根目录就宣称历史可迁移。私有启动主目录、临时数据及实例私有原生状态使用传入的 `DiscoveryOptions.RuntimeRoot`。设备凭据和连接身份必须留在保留状态之外。仅有状态路径并不意味着计算资源替换后的原生续接已通过资格验证:适配器必须保持原生所有权,在提交模型输入之前拒绝缺失或外来历史,并在另一个写入者挂载前完成原生关闭。 [`cli/agent_registration.go`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/apps/daemon/internal/cli/agent_registration.go) 遍历已发现的 Runtime,并调用 `agent/harness.go` 中的 `Registry.Register`。它验证发现过程是否保留了声明的 kind,并按以下顺序注册该 Runtime: diff --git a/contracts/agents-api/zh/index.md b/contracts/agents-api/zh/index.md index 85acfdf9f..9d55afff7 100644 --- a/contracts/agents-api/zh/index.md +++ b/contracts/agents-api/zh/index.md @@ -1,7 +1,7 @@ --- title: "Agents API 覆盖台账" source: contracts/agents-api/index.md -source_hash: f4a4bf113b88eb25ebccea95cc3125a7413e40493066e60bf2215da86471a5c8 +source_hash: 2394633ee55b7f86d374d5ab3827d0168d3dcbae8b962115063613e288b4cfda --- Core 旨在以下方固定版本为准支持完整的 OpenAI Agents API([public API rule](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/AGENTS.md#public-api))。本台账记录 Core 对各项资源实现了哪些内容、哪些契约保存其详细信息,并列出相对于 OpenAI 服务的所有已知差异和所有未解决缺口。[API namespaces and credentials](../../../docs/zh/api/index.md) 说明谁调用哪些 API;[Agents API guide](../../../docs/zh/api/public-agent-api.md) 介绍使用方法。 @@ -130,6 +130,7 @@ Core 自身字段位于 `x_agents_core` 中([Core extensions](../../../docs/zh - Runtime 不会实施 `disabled` 或 `restricted` 网络,因此需要这些网络的 Session 会被拒绝([restricted network policy](environments.md#restricted-network))。 - `packages.system` 会被拒绝;系统软件包必须预先安装。 +- 冷续接不恢复进程内存、后台进程或临时系统盘修改;已完成的 setup 不会重放。参见 [Environment 持久化](environments.md#runtime-capability-preparation)。跨节点内存恢复及接管未确认停止的旧写入者尚未通过资格验证。 **Files 和 Environment files** diff --git a/contracts/agents-api/zh/node-generation-protocol.md b/contracts/agents-api/zh/node-generation-protocol.md index 99998679f..b147594c5 100644 --- a/contracts/agents-api/zh/node-generation-protocol.md +++ b/contracts/agents-api/zh/node-generation-protocol.md @@ -1,7 +1,7 @@ --- title: "沙箱节点协议" source: contracts/agents-api/node-generation-protocol.md -source_hash: e9a9e3d1027d797bdf467d867023fa11792866ea93252f94904a66f7be45af1e +source_hash: 074df236614044bf7e2c7a3049a6acbfad4bc84f29fe248600cfbcd8c3e8bf09 --- 沙箱节点在其主机上运行 Docker 或 microsandbox Provider,并通过一个 WebSocket 与 Core 相连。Core 通过该连接发送 Provider 操作;节点针对本地 Provider 执行这些操作,并报告就绪状态、主机测量值及其持有的部署代次。Core 始终是唯一的生命周期所有者:节点绝不重试变更操作或调度工作。帧和校验器位于 [`services/core/internal/sandbox/node`](https://github.com/MiniMax-AI/OpenAgentCore/tree/main/services/core/internal/sandbox/node)(`wire.go`、`generation_wire.go`);节点用于注册和读取配置的 HTTP 路由位于[机器连接 API](machine-api.md#node-routes)。 @@ -76,9 +76,9 @@ Core 发送包含以下内容的 `request` 帧: ## 代次控制 {#generation-control} -未启用代次管理的节点只服务其登记的代次,配置固定,并且不会收到准备或保留帧。支持代次管理的节点会准备 Core 在 `welcome` 和 `heartbeat_ack` 中通告的目标代次,并在此期间继续服务其持久化的服务代次;目标代次的准备独立于服务 Provider 的就绪状态。 +未启用代次管理的节点只服务其登记的代次,配置固定,并且不会收到准备或保留帧。它可以在 `Health.Generations` 报告该精确代次;支持检查点的 Provider 必须报告,并包含已验证的检查点资格。`provider_ready` 必须与报告的状态一致。支持代次管理的节点会准备 Core 在 `welcome` 和 `heartbeat_ack` 中通告的目标代次,并在此期间继续服务其持久化的服务代次;目标代次的准备独立于服务 Provider 的就绪状态。 -节点的 `hello` 和心跳最多携带八条代次观察记录。每条记录指定一个正值的有符号 64 位代次编号、其小写 SHA-256 规范摘要、`ready`、`preparing` 或 `failed` 状态,以及可选的固定诊断信息。目标代次和服务代次的记录排在前面,其余记录公平轮换。八条记录限制的是单条消息,而不是节点可保留的代次数量。省略某条观察记录绝不会授权删除,也不会暗示不存在。 +节点的 `hello` 和心跳最多携带八条代次观察记录。每条记录指定一个正值的有符号 64 位代次编号、其小写 SHA-256 规范摘要、`ready`、`preparing` 或 `failed` 状态,以及可选的固定诊断信息。目标代次和服务代次的记录排在前面,其余记录公平轮换。八条记录限制的是单条消息,而不是节点可保留的代次数量。省略某条观察记录绝不会授权删除,也不会暗示不存在。 支持检查点的 ready 代次携带 `checkpoint`,包含非空且不透明的 `artifact_domain` 和 `execution_class` token,各最多 256 字节。不支持的 Provider 省略该字段;preparing 或 failed 代次不声明它。Core 将声明与已认证连接和代次一起持久化,仅在当前在线连接的该代次仍 ready 时使用。[Provider 契约](../../../docs/zh/sandbox-provider.md#checkpoint-transfer) 定义归档与执行兼容性。 保留使用独立且有界的交换。`retention` 请求最多指定八个本地 `(generation, specification_digest)` 引用、一个 UUID、一个每次递增 1 的 `sequence`、当前 `connection_id` 和所有者 epoch。`retention_ack` 必须与完整的待处理请求匹配,包括条目顺序和身份信息,并为每个条目给出显式布尔值 `keep`。每条连接只能有一个交换处于待处理状态,断连会将其丢弃。任何未经请求、重放、过期、不完整或混合的确认都不会删除任何内容。保留流量从不使用 Provider 请求队列。 @@ -126,6 +126,6 @@ Runtime 字节缺失时,绝不将固定的放置实例迁移到当前 Runtime 准备诊断使用固定的类型化原因。只有制品传输、校验和或版本来源验证失败才会报告 `runtime_download_failed`;私有准备器通过退出类别指示这一类失败,Core 和节点都不解析 stderr。Provider 故障、所有权故障、取消和未分类故障保留其类型化代码,或使用 `provider_unavailable`。协议中不会传输任何 Provider 原始文本。 -创建操作通过 `Bootstrap.Harness` 将会话选择传递到 Runtime 启动协议版本 2。Core 和节点使用协议版本 7,需协调升级配套组件。`Bootstrap.Harness` 是必填字段。十项暂停操作均属于同一个 `SandboxProvider`,必须完整声明支持或不支持;不通过可选接口分派。`Observe` 每次只观测一个 allocation。 +创建操作通过 `Bootstrap.Harness` 将会话选择传递到 Runtime 启动协议版本 2。Core 和节点使用协议版本 8,需协调升级配套组件。`Bootstrap.Harness` 是必填字段。十项暂停操作均属于同一个 `SandboxProvider`,必须完整声明支持或不支持;不通过可选接口分派。`Observe` 每次只观测一个 allocation。 当前线协议版本为 7。Create 引导和 Resume 请求可携带可选的工作区文件系统绑定。节点在转发前校验其租户及 Environment 与分配引用一致、ObjectID 不可变且有效,以及挂载配置 ID 与所传配置一致。文件系统解析器负责适配器原生所有权校验。未提供绑定时选择自有存储;已提供绑定时不得回退。错误响应仅可保留原生 ID 非空且名称、代次、保留状态来源均匹配请求的 Resume 目标;这仅为清理证据,不代表恢复成功。 diff --git a/contracts/agents-api/zh/sandbox-deployment.md b/contracts/agents-api/zh/sandbox-deployment.md index 51a9346bb..dd7f1a65c 100644 --- a/contracts/agents-api/zh/sandbox-deployment.md +++ b/contracts/agents-api/zh/sandbox-deployment.md @@ -1,7 +1,7 @@ --- title: "沙箱部署" source: contracts/agents-api/sandbox-deployment.md -source_hash: ddf10342dfb4b27c1d618c07b085b64b20a1483b4cfbb09b668fae02b66a08c7 +source_hash: dffb0f8a29c36f602a2965cb45e580793e335cda90b335f9de245701c61cc476 --- 沙箱部署为 Core 管理的 `openai_hosted` 执行选择 Sandbox Provider、每个沙箱的资源以及不可变的 Runtime 发行版。PostgreSQL 为每个安装维护一个当前有效选择;Web 和 Core API 写入同一配置。节点文件保存其已安装副本和特定于主机的路径,且不能覆盖其资源或 Runtime。该选择独立于 Harness。部署可以保持未配置状态,没有节点;此时它拒绝托管准入。 @@ -125,7 +125,7 @@ POST 会在持久保存候选配置之前对其进行验证,并且不会创建 ## 代次所有权与推出 {#generation-ownership-and-rollout} -`runtime_deployment` 保存当前 specification。被取代的行仅保留不可变的 specification、构建和端点元数据,绝不保留另一个 E2B 密钥。E2B 分配在预留时绑定其代次;节点放置在 Session 准入时绑定,其分配会复制该代次,即使之后发生更新也是如此。检查、续期、命令和清理使用分配的原始 specification 和端点以及当前密钥;缺失代次绝不会回退到当前 specification。已释放代次的标识仍会保留。 +`runtime_deployment` 保存当前 specification。被取代的行仅保留不可变的 specification、构建和端点元数据,绝不保留另一个 E2B 密钥。E2B 分配在预留时绑定其代次;节点放置在调度器预留兼容容量时绑定,其分配会复制该代次,即使之后发生更新也是如此。检查、续期、命令和清理使用分配的原始 specification 和端点以及当前密钥;缺失代次绝不会回退到当前 specification。已释放代次的标识仍会保留。 当一个代次为当前代次、被尚未释放的分配或放置引用,或者被尚未移除的节点固定时,该代次会保留。节点的持久服务固定状态可跨离线时段和零资源状态保留,并与当前就绪状态相互独立。回收操作与更新和准入共享部署锁,每轮最多删除 32 个符合条件的代次行。只有确认释放后,重置才会停用固定状态并清除被取代的行。 @@ -145,7 +145,7 @@ POST 会在持久保存候选配置之前对其进行验证,并且不会创建 每个节点会添加 `rollout: {state, ready_generation, diagnostic?}`,其中 `ready_generation` 是可为 null 的持久服务固定状态,`diagnostic` 是目标代次的固定代码;分配项会添加 `deployment_generation`。仅当 `rollout.state` 为 `preparing` 或 `reset` 非 null 时,才每五秒轮询一次;旧 Session 以及失败、需要更新或离线的节点本身均不会使轮询保持活动状态。 -新的准入操作会先按在线状态、精确的服务代次就绪情况、地址和共享容量筛选节点,再优先选择最新的合格固定状态,因此最新的节点已满时不会掩盖仍有空闲资源的较旧节点。没有候选项时,准入操作不会创建临时 Session 或放置:在线节点若确实正在准备且有空闲容量,则返回 503 `sandbox_nodes_preparing`;节点集群全部已满或离线,则返回 `runtime_node_unavailable`。 +节点模式创建先验证目标部署,提交 Session、pending Environment 及初始输入,但不预留计算资源。节点全部已满、离线或准备中时,已接受的工作继续等待;未配置部署、reset 或不支持的组合仍拒绝准入。共同调度器对尚未放置的需求进行有界循环扫描,分页按需求记录时间和 Environment ID 排序,每轮使用固定时间上界,避免持续的新需求阻止重新访问较早的工作。它检查在线状态、精确服务代次就绪情况、地址、共享容量和所选代次的 Harness/文件系统兼容性,然后优先选择最新的合格固定代次。预留后的 placement 保持不可变,直到确认释放或已结清的检查点转移。有界扫描跳过暂时不可调度的需求,并在重启后继续。已有 suspended allocation 按[检查点转移契约](../../../docs/zh/sandbox-provider.md#checkpoint-transfer),通过同一容量锁预留符合条件的检查点容量;这不构成热恢复和未放置工作之间的全局公平性保证。输入保留[原始五分钟期限](./environments.md#reservations),等待容量的时间也计入其中。 ## 重置 {#reset} diff --git a/deploy/e2b/README.md b/deploy/e2b/README.md index 0da9f2e1f..8e675dc1e 100644 --- a/deploy/e2b/README.md +++ b/deploy/e2b/README.md @@ -17,8 +17,14 @@ Both are required. For the SandBase endpoint, set `E2B_API_URL` to `https://sand The E2B key is available only to the settings check and template-build steps. It is temporarily written to a private file, removed on exit, excluded from Runtime images and build reports, and never sent to a model provider. No model credentials are needed to build the template. A run creates a cloud template build and may incur the provider's build charges; do not trigger a run merely to validate workflow syntax. +## Bounded archive uploads + +The workflow splits the Runtime archive into files of at most 32 MiB and uploads each through the SDK's standard template file-copy API. These are independent files, not a multipart upload extension. The template concatenates them in order, verifies the complete archive's SHA-256, then removes the parts before the maintained extraction step. It still produces one template with unchanged Runtime contents. Archives are bounded to 8 GiB. + +For an upload failure before build submission, inspect the original run's logs and use `resume_build` with its `templateID:build_UUID` and the same explicit source SHA in `ref`. Recovery checks that the authenticated build is still `waiting` with no logs before reusing its identifiers. Do not use recovery after build submission or an ambiguous request outcome. All other runs create a new template build. + ## Output and qualification -The run summary and the `e2b-template-*` artifact contain `template.json`: the immutable `templateID:build_UUID`, source commit and branch, image ID, packaged Runtime checksum, base image, endpoint and advertised Harnesses. Template names include the run ID and attempt so another run does not replace this build's name. There is no automatic retry of an uncertain cloud build; inspect the provider before starting another run after a failure. +The successful run summary and the `e2b-template-*` artifact contain `template.json`: the immutable `templateID:build_UUID`, source commit and branch, image ID, packaged Runtime checksum, base image, endpoint and advertised Harnesses. Template names include the run ID and attempt so another run does not replace this build's name. The artifact also retains `upload.json` (chunk count and archive checksum) and `build-request.json` (cloud identifiers and source revision) when those stages are reached, including on failure. There is no automatic retry of an uncertain cloud build; inspect the provider before starting another run after a failure. The image checks verify committed daemon/adapter identity and native package loading/version checks. Template build completion establishes packaging readiness, not Core enrollment, real model execution or pause/resume qualification. The workflow changes no active Core selection, Kubernetes resource or production Session. [Sandbox deployment](../../contracts/agents-api/sandbox-deployment.md) owns later template selection and generation behavior. diff --git a/deploy/node/node_generations.py b/deploy/node/node_generations.py index 70e61fe0d..d056421f0 100644 --- a/deploy/node/node_generations.py +++ b/deploy/node/node_generations.py @@ -239,6 +239,9 @@ def validate_preparation_plan(root, plan, base, installer): if not any(paths == [release / name for name in installer.MICRO] for release in (root, root / "releases" / source)): raise installer.InstallError("Preparation artifacts are outside their immutable release") expected_home = generation_home(root, plan, base, installer) + if micro["checkpoint_root"] != base["native"]["checkpoint_root"]: + raise installer.InstallError("Preparation checkpoint store differs") + installer.prepare_checkpoint_root(Path(micro["checkpoint_root"]), initialize=False) if Path(micro["runtime_home"]) != expected_home: raise installer.InstallError("Preparation native store differs") for path in paths + [expected_home]: @@ -399,6 +402,7 @@ def prepare(args, installer): raise installer.RuntimeDownloadError("Runtime release provenance differs") from error if args.provider == "microsandbox": args.runtime_home = Path(value["native"]["runtime_home"]) if value else generation_home(root, args.configuration, base, installer) + args.checkpoint_root = Path(base["native"]["checkpoint_root"]) if value is None: value = installer.provider_config(root / "releases" / runtime["source_commit"], args, runtime["image_id"]) if preparation is None: diff --git a/deploy/node/node_install.py b/deploy/node/node_install.py index d563a29cf..5cda1e2fc 100644 --- a/deploy/node/node_install.py +++ b/deploy/node/node_install.py @@ -279,6 +279,49 @@ def micro_home(installation_id): return directory +def checkpoint_root(args): + return getattr(args, "checkpoint_root", None) or Path.home() / ".oac/checkpoints" / hashlib.sha256(args.installation_id.encode()).hexdigest()[:12] + + +def prepare_checkpoint_root(directory, *, initialize=True): + directory = Path(directory) + if not directory.is_absolute() or directory.resolve() != directory: + raise InstallError("Checkpoint root must be a canonical absolute directory") + if not initialize and not directory.is_dir(): + raise InstallError("Retained checkpoint root is missing") + safe_directory(directory) + marker = directory / ".oac-checkpoint-store" + if not existing_file(marker): + if not initialize: + raise InstallError("Retained checkpoint store identity is missing") + if any(directory.iterdir()): + raise InstallError("Checkpoint root contains unowned state") + try: + descriptor = os.open(marker, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600) + except FileExistsError: + pass + else: + with os.fdopen(descriptor, "w") as output: + output.write(str(uuid.uuid4()) + "\n") + output.flush() + os.fsync(output.fileno()) + existing_file(marker) + value = marker.read_text() + try: + identity = uuid.UUID(value.rstrip("\n")) + except ValueError: + raise InstallError("Invalid checkpoint store identity") from None + if identity.int == 0 or value != str(identity) + "\n": + raise InstallError("Invalid checkpoint store identity") + with marker.open("rb") as source: + os.fsync(source.fileno()) + descriptor = os.open(directory, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + os.fsync(descriptor) + finally: + os.close(descriptor) + + def provider_config(root, args, runtime_image): result = {"installation_id": args.installation_id, "provider": args.provider, "core_url": args.core_url + "/api/v1", "specification": args.configuration["specification"], "generation": args.configuration["generation"]} @@ -294,6 +337,7 @@ def provider_config(root, args, runtime_image): result["native"] = { "helper_path": str(root / MICRO[0]), "runtime_path": str(root / MICRO[1]), "firmware_path": str(root / MICRO[2]), "runtime_home": str(getattr(args, "runtime_home", micro_home(args.installation_id))), + "checkpoint_root": str(checkpoint_root(args)), "network": {"default_egress": "deny", "default_ingress": "deny", "rules": core_rules + [ {"action": "allow", "direction": "egress", "destination": "public"}, {"action": "allow", "direction": "egress", "destination": "host", "protocol": "udp", "port": "53"}, @@ -348,8 +392,17 @@ def configure_node(root, args, token): args.configuration = node_spec.fetch(args, token, retained, open_request, allow_enrollment=not (root / "registered.json").exists()) args.provider = args.configuration["provider"] preflight(args.provider) + if args.provider != "microsandbox" and getattr(args, "checkpoint_root", None) is not None: + raise InstallError("Checkpoint root applies only to microsandbox nodes") if args.provider == "microsandbox": + retained_provider = private_json(root / "provider.json") + if retained_provider: + retained_root = Path(retained_provider["native"]["checkpoint_root"]) + if getattr(args, "checkpoint_root", None) not in (None, retained_root): + raise InstallError("Retained checkpoint root differs; preserve its state") + args.checkpoint_root = retained_root runtime_home = micro_home(args.installation_id) + prepare_checkpoint_root(checkpoint_root(args), initialize=retained_provider is None) safe_directory(runtime_home) owner = runtime_home / "oac-installation.json" if not owner.exists() and any(runtime_home.iterdir()): @@ -1244,6 +1297,7 @@ def main(argv=None): parser.add_argument("--core-url", type=origin) parser.add_argument("--provider", choices=("docker", "microsandbox"), help="Optional assertion; Core owns provider selection") parser.add_argument("--installation-id", required=True) + parser.add_argument("--checkpoint-root", type=Path, help="Private microsandbox checkpoint directory, shared across compatible nodes when configured") parser.add_argument("--enrollment-token-stdin", action="store_true", help="Read the one-time enrollment token from standard input") parser.add_argument("--generation-action", choices=("prepare", "collect"), help=argparse.SUPPRESS) parser.add_argument("--generation", type=int, help=argparse.SUPPRESS) @@ -1272,7 +1326,7 @@ def main(argv=None): if os.geteuid() != 0: raise InstallError("Node installation and removal require root. Run this command with sudo.") if args.uninstall: - if args.source_url or args.bundle or args.core_url or args.provider or args.enrollment_token_stdin: + if args.source_url or args.bundle or args.core_url or args.provider or args.enrollment_token_stdin or args.checkpoint_root: parser.error("--uninstall takes only --installation-id and --force") uninstall_system(args) return diff --git a/deploy/node/test_node_generations.py b/deploy/node/test_node_generations.py index c83ff6f0d..0b2c66311 100644 --- a/deploy/node/test_node_generations.py +++ b/deploy/node/test_node_generations.py @@ -174,7 +174,7 @@ def micro_fixture(self): runtime.chmod(0o700) image = self.value["specification"]["runtime"]["microsandbox_ref"] self.value["specification"]["runtime"]["runtime_sha256"] = hashlib.sha256(runtime.read_bytes()).hexdigest() - self.value["native"] = {"helper_path": str(self.release / "helper"), "runtime_home": str(home), "runtime_path": str(runtime), "firmware_path": str(self.release / "firmware")} + self.value["native"] = {"helper_path": str(self.release / "helper"), "runtime_home": str(home), "checkpoint_root": str(self.root / "checkpoints"), "runtime_path": str(runtime), "firmware_path": str(self.release / "firmware")} self.args.specification_digest = node_spec.digest("microsandbox", self.value["specification"]) node_generations.atomic_json(self.root / "provider.json", self.value) # This fixture changes provider before any helper exists. diff --git a/deploy/node/test_node_install.py b/deploy/node/test_node_install.py index b22168d9b..5e4c30169 100644 --- a/deploy/node/test_node_install.py +++ b/deploy/node/test_node_install.py @@ -485,7 +485,7 @@ def test_microsandbox_imports_image_and_allows_only_explicit_private_core_endpoi self.install() config = json.loads((self.root / "provider.json").read_text())["native"] # Resources, the image and artifact hashes are read from the specification. - self.assertEqual(set(config), {"helper_path", "runtime_path", "firmware_path", "runtime_home", "network"}) + self.assertEqual(set(config), {"helper_path", "runtime_path", "firmware_path", "runtime_home", "checkpoint_root", "network"}) rules = config["network"]["rules"] self.assertIn({"action": "allow", "direction": "egress", "destination": "172.29.144.1", "protocol": "tcp", "port": "24443"}, rules) self.assertNotIn("private", [rule["destination"] for rule in rules]) @@ -1117,6 +1117,35 @@ def test_microsandbox_short_home_is_stable_and_rejects_long_user_home(self): installer.micro_home("94be54a1-138c-4f30-bc87-b13686272dbe") +class CheckpointRootTests(unittest.TestCase): + def test_identity_is_private_stable_and_not_recreated(self): + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) / "checkpoints" + installer.prepare_checkpoint_root(root) + marker = root / ".oac-checkpoint-store" + first = marker.read_bytes() + installer.prepare_checkpoint_root(root) + self.assertEqual(first, marker.read_bytes()) + self.assertEqual(stat.S_IMODE(root.stat().st_mode), 0o700) + self.assertEqual(stat.S_IMODE(marker.stat().st_mode), 0o600) + marker.write_text("incomplete") + with self.assertRaises(installer.InstallError): + installer.prepare_checkpoint_root(root) + + def test_foreign_contents_and_symlink_are_not_adopted(self): + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) / "checkpoints" + root.mkdir() + (root / "unowned").write_bytes(b"data") + with self.assertRaisesRegex(installer.InstallError, "unowned"): + installer.prepare_checkpoint_root(root) + (root / "unowned").unlink() + marker = root / ".oac-checkpoint-store" + marker.symlink_to(Path(temporary) / "missing") + with self.assertRaises(installer.InstallError): + installer.prepare_checkpoint_root(root) + + class UnsupportedNodeUpdateTests(unittest.TestCase): def test_update_refuses_without_host_operations(self): with self.assertRaisesRegex(installer.InstallError, "not supported;.*reinstall"): diff --git a/docs/architecture.md b/docs/architecture.md index 4159953a7..96351610d 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -2,7 +2,9 @@ title: "Architecture" --- -OpenAgentCore separates orchestration, compute and native execution. Core owns the API and durable state. Sandbox Providers manage compute. Independent workspace filesystem adapters manage durable Environment storage through a separate protocol. A Runtime daemon prepares an Environment and runs the selected Harness, whose native SDK or protocol owns the model and tool loop. +OpenAgentCore is an Agent execution platform that brings Harnesses, models, tools, Sessions and execution environments together through replaceable protocol implementations. It separates orchestration, compute and native execution. Core owns the API and durable state. Sandbox Providers manage compute. Independent workspace filesystem adapters manage durable Environment storage through a separate protocol. A Runtime daemon prepares an Environment and runs the selected Harness, whose native SDK or protocol owns the model and tool loop. + +The recommended deployment choices are E2B-managed compute and self-managed microsandbox with independent persistent workspace storage. Both use the same Core orchestration and capability checks. Docker remains an available Provider. Compute and filesystem adapters declare their supported combinations; a deployment choice does not introduce a separate Agent execution flow. See [Sandbox Providers](./sandbox-provider.md) and [workspace filesystem providers](./workspace-provider.md) for their contracts and supported capabilities. ```mermaid flowchart TB diff --git a/docs/configuration.md b/docs/configuration.md index 384429eb9..248af0ba8 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -74,6 +74,20 @@ An unset or empty value selects the default. Edit `.env`, then run `oac apply`. | `insecure` | `false` | `true` is required for an `http` endpoint and rejected for `https` | | `headers` | none | Request headers for the endpoint. `Host`, `Content-Length`, `Content-Type` and `Content-Encoding` are reserved | +### Runtime resource directories + +The Runtime resolves its resource directories before discovering Harness adapters. `` is the Runtime home selected by `OAC_RUNTIME_HOME`. `OAC_RUNTIME_STATE_DIRECTORY` defaults only when unset; an empty, relative or noncanonical explicit value is rejected. Directory settings have no secondary file or environment fallback. + +| Runtime process setting | Default | Packaged Linux images | +| --- | --- | --- | +| `OAC_RUNTIME_INITIALIZATION_DIRECTORY` | `/initialization` | `/environment/initialization` | +| `OAC_RUNTIME_PACKAGE_DIRECTORY` | `/packages` | `/environment/packages` | +| `OAC_RUNTIME_STATE_DIRECTORY` | `` | `/environment/runtime-state` | + +The state directory contains retained native Harness history and capability-installation completion records. It is separate from the declared workspace directory. Selecting a separate state directory does not relocate Runtime device credentials or connection identity from their instance-private paths in the Runtime home; they must not be copied with native history. [Harness onboarding](../contracts/agents-api/harness-onboarding.md#register-the-adapter) defines the adapter boundary, and [Environment preparation](../contracts/agents-api/environments.md) owns initialization and package behavior. + +Placing state on an independent filesystem preserves those files when compute is replaced. Native continuation still requires a qualified Harness, matching ownership and confirmed stop of the previous writer. It does not preserve process memory or arbitrary files on the compute root disk. + ## Runtime settings: Web Runtime settings live in Core's database. Change them in Web; scripts use the same Core API with the Core key. @@ -95,7 +109,7 @@ Set `OAC_SANDBOX_MAX_ACTIVE` and `OAC_SANDBOX_MAX_RETAINED` in `.env`, then run ### Independent workspace storage -The initial supported independent filesystem combination is microsandbox with the [kernel NFS adapter](./workspace-provider.md#kernel-nfs-adapter). Mount the same NFSv4.2 export on the Linux Core host and every participating node before starting their services, at the same absolute path, for example `/srv/oac-workspaces`. The operator manages the export, mount availability and service startup ordering. Use a trusted-client AUTH_SYS export with `root_squash`; do not enable `no_root_squash` or broaden permissions to make a check pass. Prepare the namespace and service principals according to the adapter's [ownership requirements](./workspace-provider.md#kernel-nfs-adapter). +The initial supported independent filesystem combination is microsandbox with the [kernel NFS adapter](./workspace-provider.md#kernel-nfs-adapter). For a new installation, prepare storage and service accounts first, expose the mount to Core, select storage, configure microsandbox, then enroll nodes. Mount the same NFSv4.2 export on the Linux Core host and every participating node before starting their services, at the same absolute path, for example `/srv/oac-workspaces`. The operator manages the export, mount availability and service startup ordering. Use a trusted-client AUTH_SYS export with `root_squash`; do not enable `no_root_squash` or broaden permissions to make a check pass. Prepare the namespace and service principals according to the adapter's [ownership requirements](./workspace-provider.md#kernel-nfs-adapter). Core's Compose service already runs as UID 65532. Before adding a node, prepare `oac-node` with the same UID, home `/var/lib/oac-node`, a nologin shell and a nonzero primary group. Its supplementary groups may contain only its primary group, `docker` and `kvm`; an existing home must belong to that account. The installer adopts this account. Without a precreated account it allocates a system UID, which need not match Core. Resolve an existing UID or account conflict as a host administration task before enrollment; changing a running account's UID is not part of storage setup. @@ -113,7 +127,7 @@ services: create_host_path: false ``` -Select storage using `PUT /core/v1/workspace-storage` with the Core key as described in the [administration API](../contracts/agents-api/admin-api.md#workspace-storage). Generate separate canonical UUIDs for the immutable configuration `id` and the namespace marker, then submit this shape with your actual values: +Select storage using `PUT /core/v1/workspace-storage` with the Core key as described in the [administration API](../contracts/agents-api/admin-api.md#workspace-storage). For first-time setup, generate separate canonical UUIDs for the immutable configuration `id` and the namespace marker, then submit this shape with your actual values. Restores retain their existing identities. Selection runs the adapter's availability, ownership and xattr checks in Core; it does not prove that every node can resolve the mount: ```json { @@ -127,7 +141,7 @@ Select storage using `PUT /core/v1/workspace-storage` with the Core key as descr } ``` -Then create or update the microsandbox deployment through the [sandbox deployment API](../contracts/agents-api/sandbox-deployment.md#routes), setting `resources.environment_disk_mib` to `0` and retaining the required compute resources, Runtime and provider configuration. A positive value requests a quota that this adapter does not enforce and is rejected. Deployment workspace requirements and each Session's attachment are derived from the selected database configuration; do not add a filesystem field to the deployment or a second storage configuration to node files. Web has no workspace storage editor. +Then create or update the microsandbox deployment through the [sandbox deployment API](../contracts/agents-api/sandbox-deployment.md#routes), setting `resources.environment_disk_mib` to `0` and retaining the required compute resources, Runtime and provider configuration. A positive value requests a quota that this adapter does not enforce and is rejected. Deployment workspace requirements and each Session's attachment are derived from the selected database configuration; do not add a filesystem field to the deployment or a second storage configuration to node files. Web has no workspace storage editor. After the filesystem and deployment are selected, use **Nodes → Add node** to enroll the prepared hosts. Connected, ready compute alone does not qualify an external filesystem; validate actual Session creation and access through each participating node before admitting application traffic. Include the independent namespace in the [stopped-write backup and restore procedure](./getting-started/operations.md#back-up). ### Node capacity @@ -199,7 +213,7 @@ Core reads its process environment. Compose interpolates `.env` into it and moun | `OAC_CREDENTIAL_KEY_FILE` | Required. `/run/oac/credential.key`: a base64-encoded random 32-byte key. Core seals stored credentials with it | | `OAC_CORE_KEY_DIGESTS_FILE` | Required. `/run/oac/core-key-digests.json`: a JSON array with the SHA-256 of the Core key | | `OAC_INSTALLATION_ID_FILE` | Required. `/run/oac/installation.id`: the installation ID, a canonical UUID. Core refuses an ID other than the one its database recorded | -| `OAC_EXECUTION_CONCURRENCY`, `OAC_DEFAULT_HARNESS`, `OAC_HARNESSES`, `OAC_WRITE_AUDIT_RETENTION`, `OAC_OAUTH_TRUSTED_ORIGINS`, `OAC_HISTORY_SETTINGS_FILE`, `OAC_LOG_LEVEL`, `OAC_LOG_FORMAT`, `OAC_LOG_ADD_SOURCE` | The matching [process settings](#settings). Web reads the three log settings too | +| `OAC_SANDBOX_MAX_ACTIVE`, `OAC_SANDBOX_MAX_RETAINED`, `OAC_EXECUTION_CONCURRENCY`, `OAC_DEFAULT_HARNESS`, `OAC_HARNESSES`, `OAC_WRITE_AUDIT_RETENTION`, `OAC_OAUTH_TRUSTED_ORIGINS`, `OAC_HISTORY_SETTINGS_FILE`, `OAC_LOG_LEVEL`, `OAC_LOG_FORMAT`, `OAC_LOG_ADD_SOURCE` | The matching [process settings](#settings). Web reads the three log settings too | | `OAC_PROVIDER_ROOT` | Absolute adapter artifact root. The Core image sets `/opt/oac`. Each adapter owns its helper paths beneath this root. Core serves self-hosted daemon installers from its `native-installers/` directory when that holds a `catalog.json`, after checking the catalog against its own release. Adapter state lives at `/state`, the data volume's [`state/`](#compose-installations) | Core logs the history file path it loads, never environment values or file contents. diff --git a/docs/getting-started/nodes.md b/docs/getting-started/nodes.md index d4890f614..aa02e573c 100644 --- a/docs/getting-started/nodes.md +++ b/docs/getting-started/nodes.md @@ -12,6 +12,10 @@ You add a node by generating a command in Web and running it on the host. The [s - **The sandbox configuration is saved.** Open **System** → **Manage sandbox configuration**, choose **Own machines**, the backend and a sandbox size, and **Save configuration**. To change a saved configuration, choose **Reset deployment** first. Every node of an installation uses that backend. - **The console can serve the node files.** Nodes download their Runtime and provider files from the console, which redirects to the release for files it does not hold, and check each file's size and SHA-256 against the release manifest. Node hosts therefore need access to the release. Without the files, Add node says *This console has no node files for …*. +For microsandbox checkpoint recovery across nodes, add `--checkpoint-root /absolute/private/shared-store` to the installer command on each participating node. Mount the same private store at those paths before installation and grant only the node service account access. The default is a private node-local store and supports recovery on that node. The checkpoint store is separate from workspace storage and must never be guest-accessible; the [checkpoint transfer contract](../sandbox-provider.md#checkpoint-transfer) owns compatibility and retention requirements. + +Microsandbox includes the host kernel and CPU profile in its checkpoint execution compatibility. An OS or CPU-profile change can therefore make retained checkpoints incompatible even on their original node. Before maintenance, keep a node with the matching execution class and the same ready generation available for those checkpoints; otherwise their requests wait under the [checkpoint retention policy](../sandbox-provider.md#checkpoint-transfer). + For microsandbox with independent workspace storage, complete the [NFS mount and service-account setup](../configuration.md#independent-workspace-storage) on this host before enrollment. The node receives the selected immutable filesystem configuration with each binding; do not author a separate node storage setting. The Core host joins like any other host: to run sandboxes on it, add it as a node. @@ -127,7 +131,8 @@ Use manual registration when you manage the node's files and service yourself in 3. Read the node configuration with the token, which does not consume it: `GET /api/v1/sandbox-node/configuration` with `Authorization: Bearer `. 4. Write a private provider file. Copy `provider`, `installation_id`, `core_url`, `generation` and `specification` from the response, and add a `native` object with the host settings of that provider. The adapter reads sandbox size, the Runtime image and artifact hashes from `specification`: - Docker: the [Docker node configuration](../configuration.md#docker-node-configuration) fields, with `host` an explicit Unix socket, `image` the local ID of the imported Runtime image and `seccomp_file` absolute. - - microsandbox: absolute `helper_path`, `runtime_path` and `firmware_path`, a `network` policy, and `runtime_home`: a private directory, which the helper creates with mode `0700` when it is missing. microsandbox places Unix sockets under it, so keep its path within 48 bytes; the installer refuses a longer one for its own nodes. + - microsandbox: absolute `helper_path`, `runtime_path` and `firmware_path`, a `network` policy, an explicit `checkpoint_root` for private checkpoint archives, and `runtime_home`: a private directory, which the helper creates with mode `0700` when it is missing. microsandbox places Unix sockets under it, so keep its path within 48 bytes; the installer refuses a longer one for its own nodes. + Before the first microsandbox registration, initialize the empty `checkpoint_root` as the node service account using the same release's source checkout: `PYTHONPATH=deploy/node python3 -c 'from node_install import prepare_checkpoint_root; prepare_checkpoint_root("/absolute/private/checkpoints")'`. This reuses the installer's exclusive UUID-marker initialization and ownership checks. Initialize a shared store once; other nodes use the existing marker. Never replace the marker or initialize over restored or nonempty unowned storage. 5. Register, then run the node under the host's service supervisor, with real absolute paths: ```sh @@ -148,7 +153,7 @@ The node connects out to Core; Core needs no SSH or Docker TCP access to the hos ## When a node host fails -A restarted node service keeps its identity and finds its existing sandboxes again. Core never replaces a missing sandbox by itself, and never moves a Session to another node: the Session's resources show as **Node disconnected** or **Sandbox resource missing** until the original host and its storage are back, or you archive the Session. A lost node state directory is a recovery incident: restore it from its [backup](./operations.md#back-up) together with the database and the provider storage, rather than registering the host again over existing resources. +A restarted node service keeps its identity and finds its existing sandboxes again. Core does not take over an active or unconfirmed writer merely because its node is unreachable or its sandbox is missing: those resources show as **Node disconnected** or **Sandbox resource missing** while ownership remains unresolved. A settled retained checkpoint can follow the [checkpoint transfer contract](../sandbox-provider.md#checkpoint-transfer), and confirmed release can permit [cold replacement](../sandbox-provider.md#cold-replacement-after-checkpoint-retention). A lost node state directory is a recovery incident: restore it from its [backup](./operations.md#back-up) together with the database and the provider storage, rather than registering the host again over existing resources. ## Troubleshooting diff --git a/docs/getting-started/operations.md b/docs/getting-started/operations.md index 525d774e4..23c7a2b51 100644 --- a/docs/getting-started/operations.md +++ b/docs/getting-started/operations.md @@ -24,6 +24,9 @@ docker compose -f ~/.oac/core/compose.yaml ps The examples use the default installation directory. On Windows, invoke the management command with `& "$HOME/.oac/core/oac.exe"` followed by the same arguments. For a custom installation directory, replace the path in each command. +After local credentials and enrollment are resolved, Runtime Harness discovery and the authenticated bootstrap HTTP request run concurrently. Both must succeed before the Runtime opens its connection or publishes capabilities. Failure cancels the sibling operation and waits for cleanup. The executor remains owned by the connection lifetime. Runtime startup changes require a rebuilt, qualified Runtime template. + + ## Runtime startup latency The daemon logs `executor preparation stage` with `stage=workspace`, the executor and Session IDs, duration in milliseconds and `success`. Codex logs `codex preparation stage` for `session_plan`, `model_catalog`, `process_spawn`, `rpc_initialize` and `verification`, with the owner trace, duration and `success`. These records contain no native error text, credentials, configuration, catalog contents or command output. An omitted conditional stage is unobserved, not zero. `session_plan` contains `model_catalog`; the executor readiness interval contains workspace preparation, the adapter stages and transport overhead. Do not add nested intervals together. A failed `rpc_initialize` includes its required child cleanup. @@ -140,19 +143,45 @@ Core records which key made each public resource write; the retention of that hi ## Back up -Back up these together; a restore needs all of them: +A recoverable backup contains one consistent, stopped-write set. A database dump or Core volume export alone does not include independent workspace storage or node compute state. Use the same release when restoring; see [installation version policy](#installation-version-policy). + +### Backup contents + +- The Docker volume `_data`, including `database/`, `secrets/` and `state/`. It holds Projects, key digests, nodes, default models, encrypted credentials and execution history, including large objects. Keep `secrets/core/credential.key` with its matching database or stored credentials cannot be decrypted. +- The installation directory, including `.env`, `compose.yaml`, Compose overrides and the management command. +- Every independent workspace namespace selected by current or retained objects, even when it is outside the installation volume. Back up its entire root, including `.oac-storage-root`, object identities, staging, live data, trash and deletion markers. Preserve numeric ownership, modes, links and extended attributes, including every `user.*` attribute. Copying only `live/data` loses lifecycle evidence and can resurrect deleted identities. [Workspace storage](../workspace-provider.md#kernel-nfs-adapter) owns the layout and ownership requirements. +- Each node's state directory, `/var/lib/oac-node/.oac/nodes//`, and its provider storage: Docker volumes or microsandbox's store. Include all compute and snapshot resources still referenced by the database. See [when a node host fails](./nodes.md#when-a-node-host-fails). + +- Each private `checkpoint_root` in its entirety, including `.oac-checkpoint-store`, archives, operation journals, lock files and tombstones, together with the matching Core database. It is separate from the native SDK store and independent workspace namespace. Preserve its identity and private ownership; the [checkpoint transfer contract](../sandbox-provider.md#checkpoint-transfer) owns its lifecycle. + +### Establish a stopped-write window + +1. Block new application input, live file access and administrative mutations at the installation's ingress. Let active execution, initialization, file operations and native cleanup finish. Stop other applications or host processes that can write to the same filesystem. +2. While Core and nodes remain available, confirm that every owned writer has stopped and all Create, Kill, snapshot cleanup and filesystem mutations have settled. An offline node, timeout, missing instance or stopped service is insufficient. A confirmed suspension stops its source compute but retains a checkpoint that must be included in the backup. For the simplest cold recovery baseline, wait until eligible retained Sessions have completed checkpoint expiry cleanup and their allocations are released; retain their filesystem objects and native history under the [cold replacement contract](../sandbox-provider.md#cold-replacement-after-checkpoint-retention). +3. Stop Core and Web, then stop the node control services to prevent new lifecycle work. Recheck the exact native resources and keep all writers stopped until every part of the backup is complete. **Stopping Core or a node service does not stop its sandboxes.** The node service uses `KillMode=process`; microVMs and other provider-managed compute may continue to run independently. +4. Flush the stopped filesystem's pending writes using the storage platform's supported procedure and take its snapshot or export. Export the Core data volume and installation directory, plus required node state and provider stores, within the same window. Stop PostgreSQL before copying its `database/` files. Record the release, backup time, component inventory and checksums together, and keep the backup outside the installation being backed up. +5. Restore service availability only after all copies finish. Bring up storage before Core and nodes, verify readiness, then reopen ingress. Do not replay uncertain native work to make a backup pass. + +There is no one-command, lossless maintenance drain. Archive and deployment reset change Session lifecycle state; they are not substitutes for a backup pause that preserves resumability. If any writer or mutation remains unknown, keep the backup incomplete and resolve that ownership first. + +For a logical database copy during the stopped-write window, leave only the database service running and use: + +```sh +docker compose -f "$HOME/.oac/core/compose.yaml" exec -T database \ + pg_dump -U agents_api agents_api > oac-backup.sql +``` + +A logical dump still needs the matching secrets, state, independent filesystems and required node resources. Docker Desktop's **Volumes** export is a way to copy a stopped data volume; it does not coordinate these other components. + +### Restore and rehearse -- the Docker volume `_data`, including its `database/`, `secrets/` and `state/` directories. It holds Projects, key digests, nodes, default models, encrypted credentials and all execution history, including large objects. A logical dump: +Keep application ingress and Core/node services stopped while restoring. Fence or shut down every old writer before exposing the restored filesystem; an unreachable old host is not a fence. A rehearsal must use an isolated installation and storage copy that cannot connect to the original nodes or share their writable namespace. - ```sh - docker compose -f "$HOME/.oac/core/compose.yaml" exec -T database \ - pg_dump -U agents_api agents_api > oac-backup.sql - ``` +Restore matching secrets and database state, the complete independent filesystem namespaces and any required node identities/provider stores before starting Core. Preserve the database's configuration and namespace UUIDs, object identities and deletion markers; do not run first-time namespace initialization over restored storage. Reestablish the configured mount paths and service UIDs, then validate the mounted filesystem, ownership and xattrs from each participating service context. [Independent workspace storage](../configuration.md#independent-workspace-storage) owns deployment setup. -- the installation directory containing `.env`, `compose.yaml` and the command. The data volume's `secrets/core/credential.key` must stay with the database, or stored credentials cannot be decrypted. -- each node's state directory on its host, `/var/lib/oac-node/.oac/nodes//`, with its provider storage: Docker volumes or microsandbox's store. See [when a node host fails](./nodes.md#when-a-node-host-fails) for restoring them. +The cold released baseline resumes only qualified native history after new compute admission. A backup containing retained checkpoints also depends on the matching native runtime, node identity, provider store and external filesystem identities. File-level restore can change inode identity and prevent strict checkpoint restore even when file bytes match. This procedure does not promise portable memory snapshots or restoration onto an arbitrary replacement node; preserve unresolved ownership rather than substituting a new sandbox. -Stop with `docker compose stop`, export the complete data volume and archive the installation directory, then `docker compose start`. Docker Desktop supports volume export from its **Volumes** view. A SQL dump alone does not include the encryption key or Provider state. +Start the database, Core and node services only after their restored dependencies are complete; keep ingress restricted while checking them. In an isolated rehearsal, verify a retained Session's files and native continuation on fresh compute where qualified, confirm that deleted objects cannot be reopened, and check that no old writer can access the restored namespace. Record the results before relying on the backup for recovery. ## Uninstall diff --git a/docs/maintainers.md b/docs/maintainers.md index a410ad2d1..c3c8f33fc 100644 --- a/docs/maintainers.md +++ b/docs/maintainers.md @@ -33,7 +33,7 @@ make build-core-distribution | `OAC_NATIVE_INSTALLER_BUILD_DIR` | Native installer catalog directory; see [Native installers](#native-installers) | | `CORE_DISTRIBUTION_BUILD_DIR` | Output directory under `~/.oac`. Default: `~/.oac/build/core-distribution` | | `CORE_DISTRIBUTION_BUILD_NETWORK` | Docker build network: `default`, `host` or `none` | -| `CORE_DISTRIBUTION_MICROSANDBOX_ARCHIVE` | Cached microsandbox release archive. Default: `~/.oac/cache/microsandbox-v0.7.2-linux-x86_64.tar.gz`, downloaded when missing | +| `CORE_DISTRIBUTION_MICROSANDBOX_ARCHIVE` | Cached microsandbox release archive. Default: `~/.oac/cache/microsandbox-v0.7.8-linux-x86_64.tar.gz`, downloaded when missing | | `CORE_DISTRIBUTION_DATABASE_IMAGE` | PostgreSQL 16 image; the default is pinned by its linux/amd64 manifest digest | The build reuses the Core, Web, Runtime, SDK and helper builders. The manifest records the commit and source tree, image config and OCI manifest digests, the Runtime OCI manifest digest, the microsandbox runtime and firmware hashes, and the size and SHA-256 of every Runtime and node artifact; native installers carry only their SHA-256 in the [catalog](#native-installers). Output is the control archive and its `.sha256`, the optional offline archive, and the versioned Runtime, node and native installer assets. Nothing is published. Rebuilding into a directory that already holds this commit's distribution is refused. @@ -59,6 +59,8 @@ The catalog records the commit, the Runtime protocol version, each archive's SHA `make build-core-distribution` builds all of these. Build one on its own to test a Harness image or a helper. Run every command from the repository root; default outputs go under `${OAC_DEV_HOME:-$HOME/.oac}/build`. +The three maintained Runtime images share a build-time adjustment to the pinned Debian login profile: an inherited, exported `PATH` is preserved, while an unset `PATH` receives Debian’s defaults. This keeps the Runtime’s initialized package and user paths available in login and non-login tools without another package-path setting. The combined image inherits the same profile. A changed upstream profile fails the build for review; custom shell startup files can still explicitly change `PATH`. + **Codex Runtime image.** Extract the official npm package `@openai/codex@0.153.4-linux-x64` under `~/.oac` (for example with `npm pack --ignore-scripts` and `tar -xzf`), then: ```sh @@ -67,7 +69,7 @@ make build-codex-runtime docker build --platform linux/amd64 -t oac-runtime:codex "${OAC_DEV_HOME:-$HOME/.oac}/build/codex-runtime" ``` -The script checks the package version, builds `oac-daemon` for Linux amd64 and prepares a context with only the daemon, the unmodified native executable, its resources and `services/core/deploy/codex/Dockerfile`. +The script checks the package version, builds `oac-daemon` for Linux amd64 and prepares a context with only the daemon, the unmodified native executable, its resources, the shared login-profile build step and `services/core/deploy/codex/Dockerfile`. **Claude Code Runtime image.** Node 20 or newer and pnpm are required. @@ -101,16 +103,16 @@ make build-e2b-provider Docker builds the Linux helper for `GOARCH=amd64` (default) or `GOARCH=arm64` with the pinned CPython and Debian 12 image. The Python dependency closure, including PyInstaller, is hash-locked in `services/core/tools/e2b-provider/requirements.lock`; no E2B account key is needed. Set `E2B_PROVIDER_BUILD_DIR` for another output directory. The build is a pure function of the helper sources, `LICENSE` and the build script, so it is cached under `~/.oac/cache/e2b-provider/` by their hash and rebuilt only when they change. The output is `oac-e2b-provider-linux-.tar.gz` with its `.sha256`; it extracts to `oac-e2b-provider/` with the executable, `_internal/`, `licenses/`, `requirements.lock` and `manifest.json`. The Core image uses that tree; the host needs a compatible glibc and CA certificates, not Python. -**microsandbox helper.** Linux only, with a C compiler: +**microsandbox helper.** Linux amd64 only, with a C compiler: ```sh make build-microsandbox-provider make check-microsandbox-provider ``` -The helper is written to `~/.oac/build/microsandbox-provider/oac-microsandbox-provider`. Its separate Go module pins the microsandbox Go SDK v0.7.2 and embeds the matching FFI library; never build production with the SDK's `microsandbox_ffi_path` tag. Core itself stays a CGO-disabled build. The helper needs glibc and runs only on nodes. +The helper is written to `~/.oac/build/microsandbox-provider/oac-microsandbox-provider`. Its separate Go module pins the official microsandbox SDK source commit. Both commands and the distribution build use `scripts/build-microsandbox-provider.py`: it stages the module dependencies, verifies the matching official FFI release checksum, fills the SDK's empty release bundle and builds with that FFI embedded. The SDK source is unchanged and no vendored binary is committed. Use this entry point rather than invoking `go build` directly; never build production with the SDK's `microsandbox_ffi_path` tag. Core itself stays a CGO-disabled build. The helper needs glibc and runs only on nodes. -**microsandbox runtime.** The distribution uses the official [v0.7.2 release](https://github.com/superradcompany/microsandbox/releases/tag/v0.7.2) archive `microsandbox-linux-x86_64.tar.gz`, SHA256 `47c223e3ef5298abf05f47ed9f87981106e400d99bb3f1d042d4d6881346b18b` (`RUNTIME_ARCHIVE_SHA256` in `scripts/core-distribution-manifest.py`). The build verifies the checksum before extracting `msb` and `libkrunfw.so.5.6.1` and records both files' hashes. The helper checks those hashes on every call and never installs or upgrades them. +**microsandbox runtime.** The distribution uses the official [v0.7.8 release](https://github.com/superradcompany/microsandbox/releases/tag/v0.7.8) archive `microsandbox-linux-x86_64.tar.gz`, SHA256 `86f9f72dc3e639c7175bc07909b4b63ce412517c1ef8a2e1921171af5682fded` (`RUNTIME_ARCHIVE_SHA256` in `scripts/core-distribution-manifest.py`). The build verifies the checksum before extracting `msb` and `libkrunfw.so.5.6.1` and records both files' hashes. The helper checks those hashes on every call and never installs or upgrades them. ### Standalone Core builds diff --git a/docs/runtime-protocol.md b/docs/runtime-protocol.md index 65e37ad54..67f00d5c8 100644 --- a/docs/runtime-protocol.md +++ b/docs/runtime-protocol.md @@ -43,6 +43,7 @@ A declaration describes what the Runtime can do. Core admits a public feature on | `local_environment`, `workspace_read_preparation`, `workspace_output_export` | The Environment type is `openai_hosted` or `self_hosted` | | `workspace_read_preparation` | An idle Files directory read needs a read-only preparation | | `native_session_recovery` | A Session with a started Turn has no recorded native Session ID | +| `retained_native_history` | The Environment uses an external workspace binding; owned workspace storage does not require this capability | | `web_search_control`, `text_verbosity` | The Harness's engine profile declares that control | | `structured_output` and `message_items` | The Agent requests `json_schema` output | | `subagent_observations` | `multi_agent.enabled` is true | @@ -55,6 +56,8 @@ A declaration describes what the Runtime can do. Core admits a public feature on Core has no admission rule for `usage` and `resume`. +`retained_native_history` means the composed Runtime and Harness can reopen the same native Session from its private retained state after confirmed native shutdown. It does not promise process memory, background processes or external connection recovery. The selected filesystem must separately provide retained storage. Core uses the previous device's persisted declaration to determine whether compute can be released without terminating the Session, and validates the replacement Runtime before admitting execution. Missing or invalid native history fails explicitly; it never authorizes a new native root or replay of prior input. + The `execution_prepare` configuration carries the opt-ins Core sets for each Run: | Field | Set by Core | @@ -109,7 +112,7 @@ Usage frames and the final usage snapshot each carry the cumulative measurement ## Preparation and execution order -Environment initialization uses `runtime_prepare` on every connection, managed or user-owned; the [Environment contract](../contracts/agents-api/environments.md#runtime-capability-preparation) owns what is prepared and when. For a file or archive, send `begin`, wait for `ready`, send ordered chunks and await each matching `received` offset, then send `commit` and await `completed`. Initialization and finalization have typed headers without file data. Validate the expected outcome, offset, size and finite error code with the shared validator. One transfer is allowed per connection. A chunk receipt confirms staged bytes, not installation; a completed commit confirms that operation, not that a later Turn ran. +Environment initialization uses `runtime_prepare` on every connection, managed or user-owned; the [Environment contract](../contracts/agents-api/environments.md#runtime-capability-preparation) owns what is prepared and when. For a file or archive, send `begin`, wait for `ready`, send ordered chunks and await each matching `received` offset, then send `commit` and await `completed`. Initialization and finalization have typed headers without file data. Every `begin` requires `budget_ms`, an integer from 1 through 1800000 carrying the remaining Core-owned initialization budget; chunks and commits omit it. The Runtime enforces that relative budget from receipt of `begin`, and Core waits no longer than its original operation deadline. The complete `begin`–`commit` staging phase has a separate two-minute limit; applying a committed operation and waiting for `completed` use the remaining initialization budget, not the staging limit. Expiry before application rejects the transfer; expiry during application reports `unknown` after local mutations stop and never authorizes replay. Validate the expected outcome, offset, size and finite error code with the shared validator. One transfer is allowed per connection. A chunk receipt confirms staged bytes, not installation; a completed commit confirms that operation, not that a later Turn ran. An execution Turn runs in five steps: @@ -123,7 +126,9 @@ A preparation reserves a per-Turn admission, not a new Executor. It carries an e Preparation and start run outside the receive loop and router lock. An admission expires five minutes after it is granted, and retries do not extend that deadline; expiry does not remove the Runtime's obligation to settle cleanup. The Runtime bounds active preparation and execution separately from idle retained resources and counts closing or uncertain resources until their cleanup succeeds. A definite `execution_prepare` rejection with `preparation_capacity` leaves the queued Turn unclaimed for the Worker to retry, including when cleanup holds the capacity; any other error or uncertain delivery authorizes no replay. The Runtime retains at most 64 admission records, and an old handle never consumes a replacement's admission. These records are connection-local, not durable input replay. -Idle expiry of an Executor is a Runtime resource policy, separate from Core's active-Turn concurrency. On shutdown the Runtime closes active and idle Executors, keeps any target whose close failed and allows a later serialized retry. An ordinary disconnection closes the failed transport and keeps the exact router until shutdown succeeds; a wait timeout or failed cleanup never authorizes reconnection, and process shutdown keeps waiting rather than discarding owned native resources. Workspace operations keep their binding and settlement rules across Turn boundaries and Executor closure. +Idle expiry of an Executor is a Runtime resource policy, separate from Core's active-Turn concurrency. On shutdown the Runtime closes active and idle Executors, keeps any target whose close failed and allows a later serialized retry. An ordinary disconnection closes the failed transport and keeps the exact router until shutdown succeeds; a wait timeout or failed cleanup never authorizes reconnection, and process shutdown keeps waiting rather than discarding owned native resources. For Runtime initialization, shutdown joins the synchronous apply operation and its receipt delivery before releasing the connection; an `unknown` result remains unknown and blocks further preparation on that Router, but does not prevent reconnection after local mutations have stopped. Core retains the failed initialization and never replays it. Workspace operations keep their binding and settlement rules across Turn boundaries and Executor closure. + +Suspension closes admission and drains admitted work and receipts while retaining settled idle Executors before acknowledging `environment_quiesced`. Active, preparing, invalid or otherwise unsettled owners prevent suspension. Idle expiry is stopped and its callbacks fenced throughout suspension; exact authenticated `environment_resume` restores a fresh idle interval on the same owners. Only this planned reconnect preserves the Router and native owners across sockets; ordinary disconnect, shutdown and cancellation of the connection lifecycle still require confirmed cleanup. A drain failure keeps admission closed until shutdown. Preparation after resume retains the same native identity and immutable-configuration checks, including credential conflicts; it never silently replaces a conflicting resident Executor. The Sandbox Provider remains responsible for the filesystem flush and compute-stop guarantees of its checkpoint implementation; Runtime quiescence alone proves neither those guarantees nor restoration of native RAM or external network connections. ## Active input receipts diff --git a/docs/sandbox-provider.md b/docs/sandbox-provider.md index e38727882..d96ce60ac 100644 --- a/docs/sandbox-provider.md +++ b/docs/sandbox-provider.md @@ -160,11 +160,13 @@ Node readiness binds to the exact generation, the current connection and the own ### Allocation lifecycle +Core includes the owning Session’s immutable Harness in `Bootstrap.Harness`; every provider projects it into the [Runtime bootstrap](./runtime-bootstrap.md) without choosing an implementation. + The allocation, its dedicated daemon credential digest and the exact Session binding commit atomically before `Create`, under the execution lease and the Session lock. Only a fresh allocation receipt permits `Create`; retries and a Core restart observe the same reference without replaying it or rotating the credential. An allocation is private compute ownership, separate from public Environment connection and native readiness; adapters qualify bootstrap completion, and Core never infers it from an engine or provider name. With a configured provider, the Worker scans committed pending hosted Environments that have no allocation, which covers idle Session creation and recovery after an interruption between commit and bootstrap; an existing allocation never re-enters that path. The scan is bounded and serialized by the lifecycle owner and needs no caller action. An initial reservation without a Turn leaves its Session idle, and a daemon connection is never treated as native readiness. The same scan publishes authenticated connection observations with durable generations, after verifying the exact Session and device binding and a settled bootstrap. -Between Turns, Core checks that connected, observed compute is still its Session's running allocation; the check changes nothing and never revives a cleanup request. Running compute never expires: explicit deletion and the retained-state retention authorize its cleanup. A stopped or missing container never authorizes discarding retained workspace or history. Disabling the provider stops new hosted admission and bootstrap but never blocks cancellation, function results or input retry outcomes of existing Sessions. +Between Turns, Core checks that connected, observed compute is still its Session's running allocation; the check changes nothing and never revives a cleanup request. Running compute never expires: explicit lifecycle cleanup and retained-state retention govern reclamation. A stopped or missing container never authorizes discarding retained workspace or history or selecting replacement compute. Disabling the provider stops new hosted admission and bootstrap but never blocks cancellation, function results or input retry outcomes of existing Sessions. Terminal cleanup atomically revokes the device's authority, records the Environment's failure or expiry, settles pending input and requests cancellation, and only then calls `Kill`; original input deadlines and retry outcomes are kept. Temporary provider outages, unknown Create results and stopped compute never prove a permanent failure. After public Session deletion Core keeps the allocation and marks it released only after owned compute and volume cleanup and proof that the original Create settled; an unknown creation keeps cleanup ownership even after an absence observation, and bounded scans continue to catch late resources without another `Create`. @@ -174,37 +176,60 @@ After a hosted Environment commits, Core sends a bounded hint to the lifecycle w Each registered node has one serial lifecycle worker that owns its gate, allocation and pending cursors, connections and wake hints; E2B allocations share one serial lifecycle without a node. A thin coordinator discovers nodes and shuts workers down, and never holds its map mutex during database, provider or wait operations. Workers advance independently, so a stuck provider on one online node never stalls another: lifecycle concurrency is one operation per node and grows with the node count. Offline workers stay, so their retained resources remain observable after reconnection. -Allocation scans filter by node before their 32-row page limit, and pending scans join the unreleased committed placement. Each node advances its own cursor, including past failed observations, and wraps once at the end. Direct provisioning resolves the tenant-scoped placement before entering that node's gate, and an existing allocation must agree with it; Core never chooses another node. +Allocation scans filter by node before their 32-row page limit, and pending scans join the unreleased committed placement. Each node advances its own cursor, including past failed observations, and wraps once at the end. Direct provisioning resolves the tenant-scoped placement before entering that node's gate, and an active allocation must agree with it. Only confirmed release permits a new placement; a disconnect never moves an active allocation. Before releasing the execution lease, the coordinator stops accepting work and cancels and drains every node worker and direct caller. Lease loss affects everything; ordinary provider failures stay within their node. A planned deployment drain or node retirement cancels lifecycle contexts synchronously between leased operations, through the lease gate with its five-second bound, including an active manual reconcile, and never cancels an in-flight leased query just to change configuration. A failed cancellation fence closes manager admission and reports owner failure. A failed retirement keeps the original lifecycle identity and gate until owner shutdown, and the drain barrier stays closed. Session locks, deployment capacity transactions and revision-checked receipts stay authoritative, and no external operation holds a database lock. ### Placement and capacity -Placement is automatic: the environment-to-node placement commits with Session creation and its retry identity, callers cannot choose a node, and a retry keeps its original node even while it is offline. Node capacity counts pending reservations and unresolved resources, and new placement and a suspended-to-restoring transition share a database lock. Unknown operations keep their reservations, source teardown must be confirmed before active capacity is released, and confirmed cleanup releases placement capacity. Retained ownership needs exact provider evidence: a socket path, a missing instance or an empty listing never proves cleanup or authorizes a replacement. +Placement is automatic: Session creation commits durable pending work, and the common scheduler reserves a compatible node before allocation. The [deployment contract](../contracts/agents-api/sandbox-deployment.md#generation-ownership-and-rollout) owns waiting, ordering and generation selection. Callers cannot choose a node, and an active allocation keeps its original node even while it is offline. After confirmed allocation release, eligible retained Sessions can reserve compatible capacity again, including on another node. Node capacity counts pending reservations and unresolved resources, and new placement and a suspended-to-restoring transition share a database lock. Unknown operations keep their reservations, source teardown must be confirmed before active capacity is released, and confirmed cleanup releases placement capacity. Retained ownership needs exact provider evidence: a socket path, a missing instance or an empty listing never proves cleanup or authorizes a replacement. -The deployment's CPU, memory and disk settings, `max_active`, `max_retained` and the retained-state retention bound each node. There is no node-level drain, cross-node Session migration, multi-active Core, autoscaling or snapshot replication. Node removal is refused while the node holds allocations, retained states, reservations, unknown results or cleanup, and offline ownership is kept. +The deployment's CPU, memory and disk settings, `max_active`, `max_retained` and the retained-state retention bound each node. Cold replacement after confirmed release does not provide node-level drain, failover from an unreachable writer, concurrent writers, multi-active Core, autoscaling or snapshot replication. Node removal is refused while the node holds allocations, retained states, reservations, unknown results or cleanup, and offline ownership is kept. ### Suspension The single `SandboxProvider` contract declares `Initial`, `NewCompute`, `GetCompute`, `RenewCompute`, `Suspend`, `Resume`, `KillCompute`, `DeleteRetained`, `RunCommandCompute` and `ResumeCompute` together: all are supported or all are unsupported. No optional suspension interface or provider-name branch selects this lifecycle. -`RetainedState` is an opaque adapter-owned envelope with `Reference`, `ID`, `Data`, `OperationID`, `SourceGeneration`, `SourceName` and `SourceID`. `Data` is nonempty and at most 64 KiB. Core preserves the envelope unchanged and validates operation and source ownership; the immutable allocation provider/generation binding selects the only adapter allowed to interpret it. A retained state need not be an independent snapshot. Microsandbox retains its complete native snapshot identity in `Data`; E2B retains its versioned native pause receipt and cannot promise a separate disk image. Each adapter validates its native payload and ownership before any effect. +`RetainedState` is an opaque adapter-owned envelope with `Reference`, `ID`, `Data`, `OperationID`, `SourceGeneration`, `SourceName`, `SourceID` and optional `Compatibility`. `Data` is nonempty and at most 64 KiB. Core preserves the envelope unchanged and validates operation and source ownership; the immutable allocation provider/generation binding selects the only adapter allowed to interpret it. A retained state need not be an independent snapshot. Microsandbox retains its complete native snapshot identity in `Data`; E2B retains its versioned native pause receipt and cannot promise a separate disk image. Each adapter validates its native payload and ownership before any effect. -`Compute.RestoredFrom`, `ComputeState.Retained`, `ResourcesReleased`, `SuspendSettled` and recovery's `ReconcileOnly` are authoritative settlement evidence. A missing resource is not proof that an unknown suspension settled. `RenewCompute` renews only the exact incarnation; an uncertain result never authorizes a second create, capture or restore. Durable `runtime_compute` uses `protocol_version: "1"`; startup rejects missing or unknown versions. Existing version-1 retained envelopes remain readable without a rewrite. Snapshot-only durable shapes are not implicitly converted, and deployment never performs synthetic replay or discards retained state. +`Compute.RestoredFrom`, `ComputeState.Retained`, `ResourcesReleased`, `SuspendSettled` and recovery's `ReconcileOnly` are authoritative settlement evidence. A missing resource is not proof that an unknown suspension settled. `RenewCompute` renews only the exact incarnation; an uncertain result never authorizes a second create, capture or restore. Durable `runtime_compute` uses `protocol_version: "1"`; startup rejects missing or unknown versions. Version-1 envelopes matching the current retained-state field shape remain readable without a rewrite. Snapshot-only durable shapes are rejected even when labeled version 1; equal version numbers do not establish compatibility. Upgrade performs no migration, synthetic replay or retained-state discard. `ResumeRequest.Workspace` carries the allocation's independently owned filesystem binding. Core obtains a ready binding before restore. Suspension must prove the source released active execution before a target can become a writer. Adapters validate the binding before native effects; an adapter without independent filesystem support rejects a non-null binding. `DeleteRetained` deletes only adapter-owned retained compute state, never the workspace. Explicit Session deletion first settles compute cleanup and then deletes independent storage, preserving a single active writer throughout recovery. +Core suspends the idle work of every provider that declares suspension support, with one shared policy: normally it suspends work idle for 5 minutes (300 seconds) and keeps retained state for 24 hours (86400 seconds). The deployment's [`suspension`](../contracts/agents-api/sandbox-deployment.md#safe-response) reports these values. Core suspends initialized Environments, including those with no Turn yet, when no root or Subagent Turn is queued, in progress or waiting, no input, file operation or initialization is pending, and real activity has been idle for that time. The idle clock starts no earlier than the allocation entering the running compute phase, including after a wake. For allocations Core records initialization completion and root or child terminal transitions with the database clock in the same transaction. Candidate filtering and the Session-locked recheck compare elapsed database time with the idle duration, and the initial retained-state retention deadline is anchored to the same database observation, so Core and database host clocks need not agree. Native completion timestamps stay unchanged in public history but never drive idle admission, and heartbeats never reset activity. Before acknowledging a planned suspension, the daemon closes admission and drains native cleanup, output receipts and file work. -Core suspends the idle work of every provider that declares suspension support, with one fixed policy: it suspends work idle for 5 minutes (300 seconds) and keeps retained state for 24 hours (86400 seconds). The deployment's [`suspension`](../contracts/agents-api/sandbox-deployment.md#safe-response) reports these values. Core suspends initialized Environments, including those with no Turn yet, when no root or Subagent Turn is queued, in progress or waiting, no input, file operation or initialization is pending, and real activity has been idle for that time. For allocations Core records initialization completion and root or child terminal transitions with the database clock in the same transaction. Candidate filtering and the Session-locked recheck compare elapsed database time with the idle duration, and the initial retained-state retention deadline is anchored to the same database observation, so Core and database host clocks need not agree. Native completion timestamps stay unchanged in public history but never drive idle admission, and heartbeats never reset activity. Before acknowledging a planned suspension, the daemon closes admission and drains native cleanup, output receipts and file work. +Capacity pressure can suspend an already idle node allocation before the normal timeout, after a 15-second inactivity grace. The Session-locked decision rechecks pending work, wake requests and activity, then takes the shared deployment capacity lock and confirms compatible waiting demand. It never interrupts an active Turn or releases capacity before the ordinary quiesce, capture and source-stop proofs complete. Pressure-triggered reclamation is conservative: an in-flight reclamation delays another only when its node is currently eligible to serve the same demand. Unavailable or incompatible nodes keep their ownership receipts without blocking healthy capacity. If suspension cannot make capacity usable, the waiting request keeps its original deadline. The Worker lease, the Session lock and the per-node gates own suspension for every provider. New Turn claims, file-write intents and capture admission serialize under the Session lock and share one compute-phase check; new pending work cancels a capture and wakes the same source. Normal preparation waits for the compute phase to be running, after the authenticated resume handshake, and pending input stays pending when its promotion conflicts with a lifecycle transition. Compute phases and revision-checked receipts live on the allocation. Core persists quiesce, capture and restore intent before the effect, only a fresh receipt performs a capture or restore, and recovery observes the exact attempt without retrying an unknown creation, capture or restore. Consumed retained state never rolls a running generation back. Deletion, revocation and retention expiry win over wake, up to the final database compare-and-swap, and unknown cleanup identities are kept until owned resources are confirmed absent. Consumed artifacts and old compute are deleted, so suspension cycles never build a chain of writable disks. -Queued work and live Environment file access wake a suspended Environment; history and published Artifact reads do not. Planned suspension uses an Environment and suspension token on the daemon connection. A PID and start-time fenced local control signal (`RunCommandCompute`) wakes the parked daemon, which authenticates again before admitting work. A transient disconnect before confirmation retries the same armed suspension with bounded attempts and backoff; a permanent authentication or protocol rejection closes it. Core owns the retained state's retention deadline, and the daemon has no timer for it. A lost quiesce acknowledgement may thaw the same source through explicit rollback but never authorizes capturing it. - -`Suspend` owns native resource release. Core releases active capacity only after receiving a bound retained handle, suspended state, `ResourcesReleased` and `SuspendSettled`. `ReconcileOnly` prohibits replaying the original capture or pause but permits adapter cleanup justified by durable retained evidence. An outcome without retained state permits rollback only with `SuspendSettled` and an exact source that can resume; other uncertainty retains ownership and closes admission. Core never unconditionally destroys the source after `Suspend`. +`Suspend` owns native resource release. Core releases active capacity only after receiving a bound retained handle, suspended state, `ResourcesReleased` and `SuspendSettled`. `ReconcileOnly` prohibits replaying the original capture or pause but permits adapter cleanup justified by durable retained evidence. An outcome without retained state permits rollback only with `SuspendSettled` and an exact source that can resume; other uncertainty retains ownership and closes admission. Core never unconditionally destroys the source after `Suspend`. Without retained state, `SuspendSettled` also proves durable closure of that exact source and operation against every late native dispatch. An in-process call fence, an allocation lock or current native absence alone cannot establish this proof. Both E2B and microsandbox keep a bounded adapter-private admission journal; rollback cannot erase same-generation closures. Only a verified later source generation can advance its fence. `Resume` consumes retained state once into the precommitted target. Recovery observes that same attempt. Core persists `waking`, authenticates and resumes the daemon, deletes consumed retained resources, then commits `running` and admits work. `DeleteRetained` is idempotent cleanup that preserves running compute. Cleanup failure keeps `waking` and cannot trigger another restore. `KillCompute` remains destructive; old-generation cleanup must not kill a newer active incarnation sharing its native ID. Every unreleased allocation, including a running one, consumes `max_retained`. Every reservation records the shared compute protocol version, including allocations whose compute phase is `disabled`. Activation rejects an unreleased allocation with a missing or different version; the old version must complete its normal cleanup before upgrade. Session history remains. +### Coordinated protocol upgrade + +Upgrade Core and nodes together with node wire version 8, the E2B helper with private wire version 4 and its rebuilt template containing hosted suspension control, and the microsandbox helper with private wire version 5. All Providers deliver Runtime bootstrap version 2. Each boundary validates its own contract; version numbers are not interchangeable across boundaries. Before activation, use the previous compatible release to complete normal cleanup of incompatible unreleased allocations. The startup fence preserves their stored receipts and Session history and makes no Provider calls to migrate them. + +### Checkpoint transfer + +A checkpoint-capable generation reports `CheckpointCompatibility`: opaque `ArtifactDomain` and `ExecutionClass` tokens. A transferable `RetainedState` carries both tokens in `Compatibility`. A direct retained state may omit them; omission never qualifies cross-node restoration or cleanup. E2B native pause does not declare checkpoint portability. Core compares these values without interpreting CPU features, filesystem paths or storage implementations. A restore destination must be online and ready for the allocation's immutable deployment generation, match both tokens, and have active capacity plus a retained slot when moving from another node. The allocation, Device and Session identities remain unchanged. The Session and deployment transaction commits the destination route, placement and exact restore intent before target-side native work. No capacity or compatible destination leaves the retained snapshot owned until its configured deadline. The target must already hold that exact generation; Core does not automatically prepare historical generations on new nodes. During upgrades, retain eligible nodes and their generation providers until their checkpoint retention obligations end. + +For transferable checkpoints, `Suspend` may report `ResourcesReleased` and `SuspendSettled` only after publishing a verified full archive durably, stopping the exact source compute and settling source-local capture and cleanup obligations. Core persists the bound retained state before marking it suspended; native absence alone is insufficient. `Suspend.ReconcileOnly` may finish archive publication and source cleanup for an already verified snapshot of the same operation, but cannot capture another snapshot or stop a source that has resumed running. If a previously published archive is missing or damaged, `Suspend.ReconcileOnly` may return its exact durable ownership receipt with `Status: unknown` and `ResourcesReleased: false` for cleanup; that receipt proves neither current archive usability nor source absence. `Resume` still verifies the archive before execution. After the source-settlement barrier, source-node unavailability does not prevent restoration. Deletion and ordinary expiry can transfer an idle suspended allocation's cleanup route to an online node in the same artifact domain; deletion does not require execution-class compatibility. The cleanup destination must already serve the same immutable generation and have retained capacity. Without such a node, cleanup stays pending and ownership is not released; operators must restore an eligible node or its capacity. An unknown running or restoring writer never moves merely because its node is unreachable. + +Recovery uses `Resume.ReconcileOnly` for the persisted target and operation. `RestoreAttemptClosed` is an exact operation ID, not a generic absence flag: the adapter may return it only after durably closing an attempt that never entered native execution and fencing every late request for that ID. An adapter may also close a durably admitted attempt when it can prove that its current invocation failed before dispatching native restoration. The microsandbox adapter does this only when the fresh request's wire deadline expires before native dispatch; archive corruption, qualification errors and opaque SDK failures do not authorize an automatic retry. Core then persists a new operation ID before attempting execution. Once native restoration has been dispatched, an unknown outcome retains the same target, ownership and original retention deadline; a missing native listing, helper death or timeout does not authorize replay. Such a restore can remain unavailable until explicit deletion or ordinary retention expiry, and pending inputs can reach their public deadline while waiting. User deletion still requires exact native settlement before resources and ownership can be released. An in-place adapter may retain a nonempty native ID: the closure must equal the requested ID, logical generation, name and retained provenance. The closure proves that this restore attempt was never dispatched, not that its underlying paused source is absent. + +The microsandbox adapter requires an explicit private `checkpoint_root`, outside every guest-accessible filesystem. Its directory is mode `0700`; a private namespace marker identifies the actual archive store. The default installation creates a node-local private store. Cross-node restoration requires operators to mount the same private store on each participating node; mount paths may differ. Runtime state directories and workspace bindings are never used to infer this root. The adapter verifies the archive, external filesystem identity and native execution class again before restoration. Restored RAM and file descriptors do not promise continuity of external TCP connections or replay safety for application side effects. + +### Cold replacement after checkpoint retention + +Retained-state retention bounds compute resources, not the lifetime of an eligible retained Session. After the deadline, Core can clean up the allocation without terminating the Session only when its independent filesystem object is ready, initialization is complete, the Session and Environment remain nonterminal, and the previous authenticated Runtime has a persisted `retained_native_history` declaration. The capability and native recovery obligations are defined by the [Core–Runtime protocol](runtime-protocol.md). Filesystem retention alone is insufficient. + +Core confirms the original creation is settled and all owned compute and retained states are cleaned up before releasing allocation and placement ownership. The old device's authority is revoked before replacement admission. Until Create, Kill or DeleteRetained has a confirmed outcome, it keeps ownership and admits no replacement writer. A recoverable Session and Environment remain disconnected with the same filesystem object and initialization result. Archive, reset and deletion take precedence and cannot be undone by a wake request. + +Capacity pressure never shortens checkpoint retention. Until its configured deadline, a suspended allocation keeps its snapshot and retained slot; unavailable capacity or an incompatible node does not authorize cold replacement. Pending requests retain their original input deadline. After retention expires, the ordinary expiry cleanup must confirm resource deletion before another allocation can reserve that slot; files and qualified native history remain retained. + +New input or live Environment file access reserves a new allocation on a compatible node and bootstraps a fresh Runtime device. Preparation reuses the same filesystem and completed initialization rather than replaying initialization. Execution resumes the same native Session from retained history after validating the replacement Runtime's declaration. Missing or invalid native history fails explicitly; Core never substitutes a new native Session or replays completed input. History and published Artifact reads do not reserve compute. This path does not preserve VM memory, background processes or open connections, and it does not qualify an undeclared Runtime, Harness or filesystem combination. + ### Reset and archive A [reset](../contracts/agents-api/sandbox-deployment.md#reset) is durable execution state that the runtime manager advances outside its counted work. Start, escalation, cancellation, setup, update and finalization serialize through the mutation gate. Idleness is rechecked with the Session lock and then the deployment lock, never in the reverse order during finalization. Work is keyset-paged and bound to the reset's request time and generation, with the absolute deadline and validated audit provenance persisted. @@ -255,4 +280,4 @@ The node uses the explicit Unix socket in its [provider configuration](./configu `DeploymentPolicy.Workspace` declares supported attachment requirements; absence means external storage is unsupported. The derived immutable `DeploymentSpec.Workspace` receipt selects the generation mode and capabilities; its [deployment workflow](../contracts/agents-api/sandbox-deployment.md#resources) never copies filesystem configuration into the generation. Microsandbox requires `host_directory` and `user_xattr`; Docker and E2B reject external attachments. `ValidateWorkspacePolicy` uses the filesystem protocol's combination validator: an external filesystem without an enforced quota rejects a positive `EnvironmentDiskMiB`, while zero requests no quota. Root disk capacity is still required. Without an external declaration, the existing owned disk bounds apply. -Microsandbox resolves each Create and Resume binding, including observation of an interrupted restore, and passes only the local directory and immutable ObjectID to its private helper. The helper binds the whole `/environment`, including workspace, staging, initialization and packages; private HOME and Harness history remain in the VM's checkpointed root. Explicit mode, object and path labels qualify native resources; mount shape alone never selects a mode. Restore verifies the snapshot's object, remaps `/environment` through SDK `Volumes`, requires strict external mount policy and rejects every restore warning. Partial restore errors retain the exact target identity for cleanup and never establish readiness. Unknown outcomes remain observations of the original operation. Native Bind checkpoint and guest-local flock continuity evidence does not establish cross-VM fencing. +Microsandbox resolves each Create and Resume binding, including observation of an interrupted restore, and passes only the local directory and immutable ObjectID to its private helper. The helper binds the whole `/environment`; its retained contents and private native history follow the [workspace filesystem scope](workspace-provider.md#lifetime-and-filesystem-scope) and [Runtime resource directories](configuration.md#runtime-resource-directories). Private HOME, credentials and the compute root are not a durable history store. Explicit mode, object and path labels qualify native resources; mount shape alone never selects a mode. Checkpoint restore verifies the snapshot's object, remaps `/environment` through SDK `Volumes`, requires strict external mount policy and rejects every restore warning. Partial restore errors retain the exact target identity for cleanup and never establish readiness. Unknown outcomes remain observations of the original operation. Native Bind checkpoint and guest-local flock continuity do not establish fencing against an unknown writer. diff --git a/docs/workspace-provider.md b/docs/workspace-provider.md index de3ccb7e9..9900d0d26 100644 --- a/docs/workspace-provider.md +++ b/docs/workspace-provider.md @@ -41,9 +41,9 @@ Workspace access is single-writer. Core must serialize compute ownership and con ## Lifetime and filesystem scope -The object contains the whole `/environment` tree, including the workspace, staging files, initialization state and package content. They remain on the same filesystem so filesystem operations within Environment preparation preserve their semantics. A Harness's private native history/configuration remains checkpoint state; it is not redirected into the workspace object. +The object contains the whole `/environment` tree, including the workspace, staging files, initialization state, package content and the private native history of a qualified Runtime and Harness. Preparation data remains on the same filesystem so its filesystem operations preserve their semantics. Native history is private state outside the workspace directory; placement and settings are owned by [Runtime resource directories](configuration.md#runtime-resource-directories). Credentials, disposable HOME and other compute-local state have no persistence guarantee. Retained native execution requires the declared capability in the [Core–Runtime protocol](runtime-protocol.md); mounting persistent storage alone does not establish it. -Workspace storage is retained with its Session until explicit Session deletion. Archival, expiration, reset and compute checkpoint deletion do not authorize workspace deletion. Core calls `Delete` only after explicit Session deletion and confirmed compute stop; unresolved compute or storage mutations retain ownership. Failed creation does not authorize deleting storage while its Session is retained. Snapshot-expiry recovery is not defined by this contract. +Workspace storage is retained with its Session until explicit Session deletion. Archival, expiration, reset and compute checkpoint deletion do not authorize workspace deletion. Core calls `Delete` only after explicit Session deletion and confirmed compute stop; unresolved compute or storage mutations retain ownership. Failed creation does not authorize deleting storage while its Session is retained. Eligible nonterminal Sessions can survive checkpoint retention expiry and later attach the same object to fresh compute under the [cold replacement lifecycle](sandbox-provider.md#cold-replacement-after-checkpoint-retention). This does not revive archived, reset or deleted execution. ## Kernel NFS adapter @@ -55,6 +55,10 @@ Construction validates and canonicalizes JSON without filesystem I/O. `Check` ve The declaration is `host_directory`, `user_xattr: true`, `capacity_quota: false`. The native attachment contains only the namespace identity; resolution validates the complete binding against the immutable configuration and filesystem ownership before returning `objects//live/data` beneath the root. Only this data directory is exposed to the guest. Neither tenant nor application input supplies a host path. +The NFS adapter enforces no per-Environment or per-tenant byte or inode quota. One object can exhaust the shared filesystem, preventing writes to other objects and the durable terminal metadata required for safe deletion. Operators must monitor free bytes and inodes and maintain headroom for lifecycle metadata and cleanup. A capacity bound on the whole namespace does not isolate capacity between tenants or objects. + Each object owns `objects//identity`, `staging/`, `live/`, `trash/` and, after deletion begins, `deleted-marker`. The permanent identity binds the tenant, Environment and object UUIDs. Create prepares and syncs a private unique staging envelope before atomically publishing its nonempty directory as `live`; replay cannot overwrite existing live data. Delete first records a durable object-local terminal marker, retires live data to unique object-local trash and removes data before its ownership markers. Concurrent deleters converge. Minimal identity and deletion metadata remain intentionally; deletion does not mean every metadata file disappears. Create and Resolve refuse the retired identity. A late in-flight Create can leave empty controlled metadata requiring a repeated Delete after its syscall settles; it cannot authorize a new writer or reuse the object identity. Cleanup examines only that object's staging and trash, without a namespace-wide sweep or background service. -NFS storage does not supply compute fencing or automatic distributed failover. Guest `flock` is not a cross-VM writer lock. The single-writer and explicit-deletion requirements above remain mandatory. The shared filesystem preserves `/environment`, while a Harness's private native history remains in its compute checkpoint; expiration of a native snapshot can therefore remove private history even while workspace files survive. Workspace retention alone is not a promise of resumable native execution after snapshot TTL expiry. +NFS storage does not supply compute fencing or automatic distributed failover. Guest `flock` is not a cross-VM writer lock. The single-writer and explicit-deletion requirements above remain mandatory. Confirmed release can permit a fresh VM on another compatible node to attach the same object, but an offline node or an unknown Create, Kill or snapshot-cleanup result never proves that the old writer stopped. Native history and database behavior require qualification of the selected Runtime, Harness and filesystem combination; file persistence alone is not a guarantee of native Session recovery, cross-node memory restore or arbitrary crash durability. + +Core requires the selected Harness profile to support `retained_native_history` when the selected deployment generation uses external workspace storage. Creation validates that generation while holding the deployment lock, before committing the Session or compute reservation. Existing Sessions use their immutable workspace binding for Runtime admission and the final execution check. Owned workspace storage does not require this capability. diff --git a/docs/zh/architecture.md b/docs/zh/architecture.md index 4be5d241a..e1739e792 100644 --- a/docs/zh/architecture.md +++ b/docs/zh/architecture.md @@ -1,10 +1,12 @@ --- title: "架构" source: docs/architecture.md -source_hash: 75efb1d2899314719492ca4f2cd8bb5dbcdc14f54afb72ac33060a15da309d31 +source_hash: 061593c891cc4dd8b9077311bec8b8b93051c1cba9e387fb7db40a7ffc3a3de5 --- -OpenAgentCore 将编排、计算资源和原生执行分开。Core 负责 API 和持久状态。Sandbox Provider 管理计算资源。独立工作区文件系统 adapter 通过单独的协议管理 Environment 持久存储。Runtime daemon 准备 Environment 并运行选定的 Harness;Harness 的原生 SDK 或协议负责模型与工具循环。 +OpenAgentCore 是 Agent 运行平台,通过可替换的协议实现统一组织 Harness、模型、工具、Session 和执行环境。它将编排、计算资源和原生执行分开。Core 负责 API 和持久状态。Sandbox Provider 管理计算资源。独立工作区文件系统 adapter 通过单独的协议管理 Environment 持久存储。Runtime daemon 准备 Environment 并运行选定的 Harness;Harness 的原生 SDK 或协议负责模型与工具循环。 + +推荐的部署选择是 E2B 托管计算,以及 microsandbox 配合独立持久化工作区存储的自部署方案。两者使用相同的 Core 编排和能力检查。Docker 仍是可用的 Provider。计算和文件系统 adapter 声明其支持的组合;部署选择不会引入独立的 Agent 执行流程。具体契约和支持能力见 [Sandbox Provider](./sandbox-provider.md) 和[工作区文件系统 Provider](./workspace-provider.md)。 ```mermaid flowchart TB diff --git a/docs/zh/configuration.md b/docs/zh/configuration.md index 0366306c8..b46b33e0a 100644 --- a/docs/zh/configuration.md +++ b/docs/zh/configuration.md @@ -1,7 +1,7 @@ --- title: "配置参考" source: docs/configuration.md -source_hash: 1b65cd5670b441953c899b5ca1753294ed3d7bee9c4f4de6e7f0a1bfd419b640 +source_hash: 9b8fbf0b986794b5f56474668dfdee67fb39b28708c3ca0a0290b60a7dec496a --- Core 安装的每项设置都恰好只有一个归属位置,分属以下三类: @@ -78,6 +78,20 @@ Web 的 **System** 页面显示该安装的地址、默认模型和沙箱配置 | `insecure` | `false` | `http` 端点必须设为 `true`,`https` 端点不允许设为 `true` | | `headers` | 无 | 发往端点的请求标头。`Host`、`Content-Length`、`Content-Type` 和 `Content-Encoding` 为保留标头 | +### Runtime 资源目录 {#runtime-resource-directories} + +Runtime 在发现 Harness 适配器之前解析资源目录。`` 是由 `OAC_RUNTIME_HOME` 选择的 Runtime 主目录。`OAC_RUNTIME_STATE_DIRECTORY` 仅在未设置时采用默认值;显式空值、相对路径或非规范路径会被拒绝。目录设置没有其他文件或环境变量回退来源。 + +| Runtime 进程设置 | 默认值 | 打包的 Linux 镜像 | +| --- | --- | --- | +| `OAC_RUNTIME_INITIALIZATION_DIRECTORY` | `/initialization` | `/environment/initialization` | +| `OAC_RUNTIME_PACKAGE_DIRECTORY` | `/packages` | `/environment/packages` | +| `OAC_RUNTIME_STATE_DIRECTORY` | `` | `/environment/runtime-state` | + +状态目录包含保留的原生 Harness 历史和能力安装完成记录,与声明的工作区目录分开。选择独立状态目录不会将 Runtime 设备凭据或连接身份移出 Runtime 主目录中的实例私有路径;不得随原生历史复制这些数据。[Harness 接入指南](../../contracts/agents-api/zh/harness-onboarding.md#register-the-adapter)定义适配器边界,[Environment 准备](../../contracts/agents-api/zh/environments.md)负责初始化和包的行为。 + +将状态放在独立文件系统上,可以在替换计算资源时保留这些文件。原生续接仍要求 Harness 已通过资格验证、所有权匹配,并确认前一个写入者已停止。这不保留进程内存或计算资源根磁盘上的任意文件。 + ## 运行时设置:Web {#runtime-settings-web} 运行时设置存储在 Core 的数据库中。请在 Web 中修改;脚本使用同一个 Core API 和 Core 密钥。 @@ -99,7 +113,7 @@ Web 的 **System** 页面显示该安装的地址、默认模型和沙箱配置 ### 独立工作区存储 {#independent-workspace-storage} -首个支持的独立文件系统组合是 microsandbox 与[内核 NFS 适配器](./workspace-provider.md#kernel-nfs-adapter)。启动服务前,在 Linux Core 主机和所有参与节点上将同一个 NFSv4.2 导出挂载到相同的绝对路径,例如 `/srv/oac-workspaces`。运维人员负责导出、挂载可用性和服务启动顺序。使用带有 `root_squash` 的受信任客户端 AUTH_SYS 导出;不要启用 `no_root_squash` 或放宽权限来使检查通过。按照适配器的[所有权要求](./workspace-provider.md#kernel-nfs-adapter)准备命名空间和服务身份。 +首个支持的独立文件系统组合是 microsandbox 与[内核 NFS 适配器](./workspace-provider.md#kernel-nfs-adapter)。新安装应先准备存储和服务账户,再向 Core 暴露挂载、选择存储、配置 microsandbox,最后注册节点。启动服务前,在 Linux Core 主机和所有参与节点上将同一个 NFSv4.2 导出挂载到相同的绝对路径,例如 `/srv/oac-workspaces`。运维人员负责导出、挂载可用性和服务启动顺序。使用带有 `root_squash` 的受信任客户端 AUTH_SYS 导出;不要启用 `no_root_squash` 或放宽权限来使检查通过。按照适配器的[所有权要求](./workspace-provider.md#kernel-nfs-adapter)准备命名空间和服务身份。 Core 的 Compose 服务已使用 UID 65532 运行。添加节点前,创建同 UID 的 `oac-node`,主目录为 `/var/lib/oac-node`,使用 nologin shell 和非零主组。附加组只能包含其主组、`docker` 和 `kvm`;已有主目录必须属于该账户。安装器会接管此账户。如果没有预创建账户,安装器分配的系统 UID 不一定与 Core 一致。注册前应通过主机管理流程解决已有 UID 或账户冲突;修改运行中账户的 UID 不属于存储配置步骤。 @@ -117,7 +131,7 @@ services: create_host_path: false ``` -按[管理 API](../../contracts/agents-api/zh/admin-api.md#workspace-storage)说明,使用 Core key 调用 `PUT /core/v1/workspace-storage` 选择存储。分别为不可变配置 `id` 和命名空间标记生成规范 UUID,然后使用实际值提交以下结构: +按[管理 API](../../contracts/agents-api/zh/admin-api.md#workspace-storage)说明,使用 Core key 调用 `PUT /core/v1/workspace-storage` 选择存储。首次设置时,分别为不可变配置 `id` 和命名空间标记生成规范 UUID,然后使用实际值提交以下结构。恢复时保留已有标识。选择操作会在 Core 中运行适配器的可用性、所有权和 xattr 检查;它不证明每个节点均能解析该挂载: ```json { @@ -131,7 +145,7 @@ services: } ``` -随后通过[沙箱部署 API](../../contracts/agents-api/zh/sandbox-deployment.md#routes)创建或更新 microsandbox 部署,将 `resources.environment_disk_mib` 设为 `0`,并保留所需计算资源、Runtime 和 Provider 配置。正数表示请求配额,本适配器不实施此配额,因此会拒绝。部署的工作区要求和每个 Session 的挂载凭据均从已选数据库配置派生;不要向部署添加文件系统字段,也不要在节点文件中添加第二份存储配置。Web 没有工作区存储编辑器。 +随后通过[沙箱部署 API](../../contracts/agents-api/zh/sandbox-deployment.md#routes)创建或更新 microsandbox 部署,将 `resources.environment_disk_mib` 设为 `0`,并保留所需计算资源、Runtime 和 Provider 配置。正数表示请求配额,本适配器不实施此配额,因此会拒绝。部署的工作区要求和每个 Session 的挂载凭据均从已选数据库配置派生;不要向部署添加文件系统字段,也不要在节点文件中添加第二份存储配置。Web 没有工作区存储编辑器。选择文件系统和部署后,通过 **Nodes → Add node** 注册已准备的主机。计算资源已连接且就绪并不足以证明外部文件系统具备资格;准入应用流量前,通过每个参与节点验证实际 Session 创建和访问。将独立命名空间纳入[停止写入后的备份恢复流程](./getting-started/operations.md#back-up)。 ### 节点容量 {#node-capacity} @@ -203,7 +217,7 @@ Core 读取进程环境。Compose 将 `.env` 插值到环境中,并把机密 | `OAC_CREDENTIAL_KEY_FILE` | 必填。`/run/oac/credential.key`:Base64 编码的 32 字节随机密钥。Core 用它加密存储的凭据 | | `OAC_CORE_KEY_DIGESTS_FILE` | 必填。`/run/oac/core-key-digests.json`:一个包含 Core 密钥 SHA-256 的 JSON 数组 | | `OAC_INSTALLATION_ID_FILE` | 必填。`/run/oac/installation.id`:安装 ID,采用规范 UUID 格式。如果 ID 与数据库记录的 ID 不一致,Core 会拒绝它 | -| `OAC_EXECUTION_CONCURRENCY`、`OAC_DEFAULT_HARNESS`、`OAC_HARNESSES`、`OAC_WRITE_AUDIT_RETENTION`、`OAC_OAUTH_TRUSTED_ORIGINS`、`OAC_HISTORY_SETTINGS_FILE`、`OAC_LOG_LEVEL`、`OAC_LOG_FORMAT`、`OAC_LOG_ADD_SOURCE` | 对应的[进程设置](#settings)。Web 也读取三个日志设置 | +| `OAC_SANDBOX_MAX_ACTIVE`、`OAC_SANDBOX_MAX_RETAINED`、`OAC_EXECUTION_CONCURRENCY`、`OAC_DEFAULT_HARNESS`、`OAC_HARNESSES`、`OAC_WRITE_AUDIT_RETENTION`、`OAC_OAUTH_TRUSTED_ORIGINS`、`OAC_HISTORY_SETTINGS_FILE`、`OAC_LOG_LEVEL`、`OAC_LOG_FORMAT`、`OAC_LOG_ADD_SOURCE` | 对应的[进程设置](#settings)。Web 也读取三个日志设置 | | `OAC_PROVIDER_ROOT` | 适配器构件的绝对根目录。Core 镜像设置为 `/opt/oac`。每个适配器都拥有此根目录下的辅助路径。当其中的 `native-installers/` 目录包含 `catalog.json` 时,Core 在核对该目录清单与自身发行版后提供自托管守护进程安装程序。适配器状态位于 `/state`,即数据卷的 [`state/`](#compose-installations) | Core 会记录所加载的历史文件路径,但绝不记录环境变量的值或文件内容。 diff --git a/docs/zh/getting-started/nodes.md b/docs/zh/getting-started/nodes.md index d7c572de2..9e7d24e05 100644 --- a/docs/zh/getting-started/nodes.md +++ b/docs/zh/getting-started/nodes.md @@ -1,7 +1,7 @@ --- title: "添加和管理节点" source: docs/getting-started/nodes.md -source_hash: 9b75587ccab0088003e8bd12afe477af0e72fe6650045f5283700248f9df327e +source_hash: 0c62518cc740387ed03035ea4870108edeaaa82331a7509b573fb2b775ae3aba --- 节点是一台 Linux 主机,在沙箱后端为 Docker 或 microsandbox 时,为 Core 托管 Session 运行沙箱。Core 将新 Session 分配给有空余容量的节点;节点创建沙箱,沙箱回连 Core。E2B 不需要节点。应用为自己的 Session 连接的机器是[自托管执行器](self-hosted.md),而不是节点。 @@ -16,6 +16,10 @@ source_hash: 9b75587ccab0088003e8bd12afe477af0e72fe6650045f5283700248f9df327e Core 主机与其他主机一样加入:要在它上面运行沙箱,将它添加为节点。 +microsandbox 跨节点检查点恢复要求在每个参与节点的安装命令加入 `--checkpoint-root /absolute/private/shared-store`。安装前将同一个私有 store 挂载到这些路径,并只允许 node 服务账户访问。默认使用节点本地私有 store,只支持在该节点恢复。检查点 store 独立于 workspace 存储,绝不能允许 guest 访问;[检查点转移契约](../sandbox-provider.md#checkpoint-transfer) 定义兼容性和保留要求。 + +Microsandbox 的检查点执行兼容性包含主机内核和 CPU profile。因此,操作系统或 CPU profile 变化可能使保留的检查点不再兼容,甚至无法在原节点恢复。维护前,应为这些检查点保留具有匹配 execution class 且已就绪同一 generation 的节点;否则,相关请求会按[检查点保留策略](../sandbox-provider.md#checkpoint-transfer)等待。 + 对于使用独立工作区存储的 microsandbox,应在注册前完成此主机上的 [NFS 挂载和服务账户配置](../configuration.md#independent-workspace-storage)。节点随每次 binding 接收所选不可变文件系统配置;不要单独编写节点存储设置。 ## 添加节点 {#add-a-node} @@ -129,7 +133,8 @@ root 只准备账号、组和服务单元;其他操作(包括 Docker 网络 3. 使用令牌读取节点配置,不会消耗令牌:`GET /api/v1/sandbox-node/configuration`,带 `Authorization: Bearer `。 4. 写入私有提供商文件。从响应复制 `provider`、`installation_id`、`core_url`、`generation` 和 `specification`,并添加 `native` 对象,写入该提供商的主机设置。适配器从 `specification` 读取沙箱规格、Runtime 镜像和产物哈希: - Docker:[Docker 节点配置](../configuration.md#docker-node-configuration)中的字段,其中 `host` 是显式 Unix 套接字,`image` 是导入的 Runtime 镜像的本地 ID,`seccomp_file` 是绝对路径。 - - microsandbox:绝对路径 `helper_path`、`runtime_path` 和 `firmware_path`;`network` 策略;以及 `runtime_home` 私有目录。目录缺失时辅助程序以 `0700` 创建。microsandbox 在其中放置 Unix 套接字,因此路径不要超过 48 字节;安装程序对自管节点拒绝更长路径。 + - microsandbox:绝对路径 `helper_path`、`runtime_path` 和 `firmware_path`;`network` 策略;私有检查点归档的显式 `checkpoint_root`;以及 `runtime_home` 私有目录。目录缺失时辅助程序以 `0700` 创建。microsandbox 在其中放置 Unix 套接字,因此路径不要超过 48 字节;安装程序对自管节点拒绝更长路径。 + 首次注册 microsandbox 前,以 node 服务账户在相同发行版本的源码 checkout 中初始化空的 `checkpoint_root`:`PYTHONPATH=deploy/node python3 -c 'from node_install import prepare_checkpoint_root; prepare_checkpoint_root("/absolute/private/checkpoints")'`。它复用安装程序的 UUID marker 排他初始化与所有权检查。共享 store 只初始化一次,其他 node 使用已有 marker。不得替换 marker,也不得在恢复后的存储或非空且无所有权证明的目录上重新初始化。 5. 使用真实绝对路径注册,然后通过主机服务管理器运行节点: ```sh @@ -150,7 +155,7 @@ root 只准备账号、组和服务单元;其他操作(包括 Docker 网络 ## 节点主机故障时 {#when-a-node-host-fails} -重启节点服务会保留身份并重新发现已有沙箱。Core 不会自行替换缺失沙箱,也不会将 Session 移到其他节点:Session 资源显示 **Node disconnected** 或 **Sandbox resource missing**,直到原主机及存储恢复,或你归档 Session。丢失节点状态目录属于恢复事件:从[备份](operations.md#back-up)恢复,并同时恢复数据库和提供商存储;不要在已有资源上重新注册主机。 +重启节点服务会保留身份并重新发现已有沙箱。Core 不会仅因节点不可达或沙箱缺失就接管活动或未确认的 writer:所有权未结清时,这些资源显示 **Node disconnected** 或 **Sandbox resource missing**。已结清的保留检查点可以遵循[检查点转移契约](../sandbox-provider.md#checkpoint-transfer),确认释放后可以按条件进行[冷替换](../sandbox-provider.md#cold-replacement-after-checkpoint-retention)。丢失节点状态目录属于恢复事件:从[备份](operations.md#back-up)恢复,并同时恢复数据库和提供商存储;不要在已有资源上重新注册主机。 ## 问题排查 {#troubleshooting} diff --git a/docs/zh/getting-started/operations.md b/docs/zh/getting-started/operations.md index df15025fd..b2d7f7909 100644 --- a/docs/zh/getting-started/operations.md +++ b/docs/zh/getting-started/operations.md @@ -1,7 +1,7 @@ --- title: "运维" source: docs/getting-started/operations.md -source_hash: 6d20edc1d353b8c1ba4ff093b86b43ad9cbe54a2e63df18753d761650fcae6c3 +source_hash: 54ac05e0785b1a485f93edd2b2d6bbc6bf84e36a866a442d53f4e6516813ba08 --- 安装运维人员负责 Core 主机、存储和可用性。节点主机运行各自的服务;参阅[节点](nodes.md)。设置见[配置参考](../configuration.md)。 @@ -26,9 +26,10 @@ docker compose -f ~/.oac/core/compose.yaml ps 示例使用默认安装目录。Windows 上使用 `& "$HOME/.oac/core/oac.exe"` 调用管理命令,后接相同参数。使用自定义安装目录时,替换各命令中的路径。 -## Runtime 启动延迟 {#runtime-startup-latency} +本地凭据和注册解析完成后,Runtime 的 Harness 发现与已认证的引导 HTTP 请求并发执行。两者都成功后才建立连接并发布能力。失败会取消另一项操作并等待其清理。执行器仍由连接生命周期持有。Runtime 启动变更需要重新构建并验证 Runtime 模板。 + -本地凭据解析和注册绑定完成后,Runtime 的 Harness 探测与带认证的 bootstrap HTTP 请求并行执行。两项都成功后,Runtime 才建立连接并公布能力;任一失败都会取消另一项并等待其清理完成。重连和挂起继续使用原有生命周期,并行执行不会跳过可执行程序或凭据校验。 +## Runtime 启动延迟 {#runtime-startup-latency} daemon 通过 `executor preparation stage` 记录 `stage=workspace`、executor 和 Session ID、毫秒耗时及 `success`。Codex 通过 `codex preparation stage` 记录 `session_plan`、`model_catalog`、`process_spawn`、`rpc_initialize` 和 `verification`,携带所属请求的 trace、耗时及 `success`。这些记录不包含原生错误文本、凭据、配置、模型目录内容或命令输出。未执行的条件阶段表示未观测,不能按零计算。`session_plan` 包含 `model_catalog`;executor 就绪耗时包含工作区准备、adapter 各阶段和传输开销,不应重复相加。失败的 `rpc_initialize` 包含必要的子进程清理。 @@ -38,7 +39,6 @@ daemon 通过 `executor preparation stage` 记录 `stage=workspace`、executor app-server 初始化前仍会校验并固定模型目录。阶段计时用于区分目录准备与原生进程初始化,本身不证明提速。应在相同 Runtime 模板、模型和 Provider 下对比全新及复用 Session,并同时验证持久化回复、用量和首字延迟。Runtime 启动逻辑变更需要重新构建和验收 Runtime 模板;仅替换 Core 不会更新既有沙箱。 - ## 服务健康状态 {#service-health} 根据不同问题使用这些观察: @@ -144,20 +144,45 @@ Core 记录每次公开资源写入所使用的密钥;历史保留策略为 [` ## 备份 {#back-up} -一起备份这些内容;恢复时全部需要: +可恢复的备份必须是一组在停止写入期间取得的一致数据。仅数据库转储或 Core 卷导出不包含独立工作区存储及节点计算状态。恢复时使用相同版本;参见[安装版本策略](#installation-version-policy)。 + +### 备份内容 {#backup-contents} + +- Docker 卷 `_data`,包括 `database/`、`secrets/` 和 `state/`。其中保存 Project、密钥摘要、节点、默认模型、加密凭据和执行历史(含大对象)。`secrets/core/credential.key` 必须与匹配的数据库一起保留,否则存储的凭据无法解密。 +- 安装目录,包括 `.env`、`compose.yaml`、Compose 覆盖文件和管理命令。 +- 当前或保留对象所使用的每个独立工作区命名空间,即使其位于安装卷之外。备份整个根目录,包括 `.oac-storage-root`、对象标识、staging、live 数据、trash 和删除标记。保留数值所有者、权限、链接及扩展属性,包括全部 `user.*` 属性。仅复制 `live/data` 会丢失生命周期证据,并可能使已删除标识复活。[工作区存储](../workspace-provider.md#kernel-nfs-adapter)负责定义布局与所有权要求。 +- 各节点状态目录 `/var/lib/oac-node/.oac/nodes//` 及其 Provider 存储:Docker 卷或 microsandbox 存储。包含数据库仍然引用的全部计算资源和快照。参见[节点主机故障时](./nodes.md#when-a-node-host-fails)。 + +- 每个私有 `checkpoint_root` 的全部内容,包括 `.oac-checkpoint-store`、归档、operation journal、lock 文件与 tombstone,并与匹配的 Core 数据库一起保留。它独立于原生 SDK store 和工作区 namespace。保留其身份和私有所有权;[检查点转移契约](../sandbox-provider.md#checkpoint-transfer)定义其生命周期。 + +### 建立停止写入窗口 {#establish-a-stopped-write-window} + +1. 在安装入口阻止新应用 input、实时文件访问和管理变更。等待活动执行、初始化、文件操作和原生清理完成。停止其他能够写入同一文件系统的应用或主机进程。 +2. 保持 Core 和节点可用,确认每个所属写入方均已停止,且所有 Create、Kill、快照清理及文件系统变更均已结算。节点离线、超时、实例缺失或服务停止均不足以证明这一点。确认挂起会停止源计算资源,但保留的检查点也必须纳入备份。最简单的冷恢复基线是等待符合条件的保留 Session 完成检查点到期清理、allocation 已释放,再按[冷替换契约](../sandbox-provider.md#cold-replacement-after-checkpoint-retention)保留文件系统对象及原生历史。 +3. 停止 Core 和 Web,再停止节点控制服务,以防止新的生命周期操作。重新核实精确的原生资源,保持所有写入方停止,直到备份各部分全部完成。**停止 Core 或节点服务不会停止其沙箱。** 节点服务使用 `KillMode=process`;microVM 及其他由 Provider 管理的计算资源可能继续独立运行。 +4. 使用存储平台支持的流程将已停止写入的文件系统中的待写数据落盘,并创建其快照或导出。在同一窗口内导出 Core 数据卷、安装目录及所需的节点状态和 Provider 存储。复制 PostgreSQL 的 `database/` 文件前必须停止 PostgreSQL。一起记录版本、备份时间、组成清单和校验值,并将备份保存在被备份安装之外。 +5. 所有复制完成后才恢复服务。先恢复存储,再启动 Core 和节点,验证就绪后开放入口。不要通过重放不确定的原生操作来让备份通过。 + +当前没有一条命令即可完成且无损的维护排空流程。归档和部署重置会改变 Session 生命周期状态,不能替代保持续接能力的备份暂停。任何写入方或变更仍然未知时,应将备份保留为未完成状态,先解决相应归属。 + +若在停止写入窗口中进行数据库逻辑复制,仅保持 database 服务运行,执行: + +```sh +docker compose -f "$HOME/.oac/core/compose.yaml" exec -T database \ + pg_dump -U agents_api agents_api > oac-backup.sql +``` + +逻辑转储仍然需要匹配的 secrets、state、独立文件系统及所需节点资源。Docker Desktop 的 **Volumes** 导出可用于复制已停止的数据卷,但不会协调这些其他部分。 -- Docker 卷 `_data`,包括其中的 `database/`、`secrets/` 和 `state/` 目录。其中包含 Project、密钥摘要、节点、默认模型、加密凭据和全部执行历史(含大对象)。逻辑备份: +### 恢复与演练 {#restore-and-rehearse} - ```sh - docker compose -f "$HOME/.oac/core/compose.yaml" exec -T database \ - pg_dump -U agents_api agents_api > oac-backup.sql - ``` +恢复期间保持应用入口及 Core/节点服务关闭。暴露恢复后的文件系统前,必须隔离或关闭每一个旧写入方;旧主机不可达不代表已隔离。演练必须使用隔离的安装和存储副本,不能连接原节点,也不能共享其可写命名空间。 -- 安装目录中的 `.env`、`compose.yaml` 和管理命令。数据卷内的 `secrets/core/credential.key` 必须与数据库一起保留,否则存储的凭据无法解密。 +启动 Core 前,恢复相匹配的 secrets 和数据库状态、完整的独立文件系统命名空间,以及任何所需的节点标识和 Provider 存储。保留数据库中的配置和命名空间 UUID、对象标识及删除标记;不要对恢复后的存储执行首次命名空间初始化。重新建立配置中的挂载路径和服务 UID,然后从每个参与服务的上下文验证所挂载文件系统、所有权及 xattr。[独立工作区存储](../configuration.md#independent-workspace-storage)负责定义部署设置。 -- 各节点主机上的状态目录 `/var/lib/oac-node/.oac/nodes//` 及提供商存储:Docker 卷或 microsandbox 存储。恢复方法见[节点主机故障时](nodes.md#when-a-node-host-fails)。 +已释放计算资源的冷恢复基线仅在新计算资源准入后续接具备资格的原生历史。包含保留检查点的备份还依赖匹配的原生 Runtime、节点标识、Provider 存储和外部文件系统标识。文件级恢复可能改变 inode 标识,即使文件字节一致,也可能阻止严格检查点恢复。此流程不承诺可移植内存快照,也不承诺恢复到任意替代节点;应保留未解决的归属,而不是替换为新沙箱。 -运行 `docker compose stop`,导出完整数据卷并归档安装目录,再运行 `docker compose start`。Docker Desktop 的 **Volumes** 页面支持导出数据卷。SQL 转储不包含加密密钥和 Provider 状态。 +恢复依赖全部完整后才启动 database、Core 和节点服务;检查期间保持入口受限。在隔离演练中,验证保留 Session 的文件和具备资格时在新计算资源上的原生续接,确认已删除对象无法重新打开,并检查旧写入方均无法访问恢复后的命名空间。在依赖备份进行恢复前记录结果。 ## 卸载 {#uninstall} diff --git a/docs/zh/maintainers.md b/docs/zh/maintainers.md index 924c3b19f..de88bb348 100644 --- a/docs/zh/maintainers.md +++ b/docs/zh/maintainers.md @@ -1,7 +1,7 @@ --- title: "构建并发布 OpenAgentCore" source: docs/maintainers.md -source_hash: aaadeb3a5e99b8d926e9f78b7fc41c7aba808e0d2af58969119e67954f9d7a58 +source_hash: f91505a6502e09cc13b5ece7cb4ba578f6d2861621d70a1f836f89dcf9299ea4 --- 本指南面向负责构建和发布 OpenAgentCore 的维护者。要安装 Core 和 Web,请使用 [安装指南](getting-started/install.md)。安装器代码遵循的规则见 [部署](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/deploy/README.md) 和 [节点安装器](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/deploy/node/README.md);必需检查见 [CONTRIBUTING](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/CONTRIBUTING.md#required-checks)。 @@ -35,7 +35,7 @@ make build-core-distribution | `OAC_NATIVE_INSTALLER_BUILD_DIR` | 原生安装器目录;请参阅[原生安装器](#native-installers) | | `CORE_DISTRIBUTION_BUILD_DIR` | `~/.oac` 下的输出目录。默认值:`~/.oac/build/core-distribution` | | `CORE_DISTRIBUTION_BUILD_NETWORK` | Docker 构建网络:`default`、`host` 或 `none` | -| `CORE_DISTRIBUTION_MICROSANDBOX_ARCHIVE` | 已缓存的 microsandbox 发布归档。默认值:`~/.oac/cache/microsandbox-v0.7.2-linux-x86_64.tar.gz`,缺失时下载 | +| `CORE_DISTRIBUTION_MICROSANDBOX_ARCHIVE` | 已缓存的 microsandbox 发布归档。默认值:`~/.oac/cache/microsandbox-v0.7.8-linux-x86_64.tar.gz`,缺失时下载 | | `CORE_DISTRIBUTION_DATABASE_IMAGE` | PostgreSQL 16 镜像;默认值通过其 linux/amd64 清单摘要固定 | 构建过程会复用 Core、Web、Runtime、SDK 和辅助程序构建器。清单会记录提交和源代码树、镜像配置及 OCI 清单摘要、Runtime OCI 清单摘要、microsandbox 运行时和固件哈希,以及每个 Runtime 和节点构件的大小与 SHA-256;原生安装器在[目录](#native-installers)中仅记录其 SHA-256。输出包括控制归档及其 `.sha256`、可选的离线归档,以及带版本号的 Runtime、节点和原生安装器资源。此过程不会发布任何内容。如果目标目录中已包含此提交的分发包,重建会拒绝执行。 @@ -61,6 +61,8 @@ export OAC_NATIVE_INSTALLER_BUILD_DIR=OUTPUT_DIR `make build-core-distribution` 会构建以下全部内容。也可以单独构建其中一项,以测试某个 Harness 镜像或辅助程序。所有命令都必须从仓库根目录运行;默认输出位于 `${OAC_DEV_HOME:-$HOME/.oac}/build` 下。 +三个维护的 Runtime 镜像共用一个构建期调整:固定 Debian 登录 profile 会保留继承且已导出的 `PATH`,未设置 `PATH` 时仍使用 Debian 默认值。这样,登录和非登录工具都能使用 Runtime 初始化的包路径与用户路径,无需另一份包路径设置。组合镜像继承同一 profile。上游 profile 结构变化会使构建失败以便审查;自定义 shell 启动文件仍可以显式更改 `PATH`。 + **Codex Runtime 镜像。** 在 `~/.oac` 下解压官方 npm 包 `@openai/codex@0.153.4-linux-x64`(例如使用 `npm pack --ignore-scripts` 和 `tar -xzf`),然后执行: ```sh @@ -69,7 +71,7 @@ make build-codex-runtime docker build --platform linux/amd64 -t oac-runtime:codex "${OAC_DEV_HOME:-$HOME/.oac}/build/codex-runtime" ``` -该脚本会检查软件包版本,为 Linux amd64 构建 `oac-daemon`,并准备一个仅包含守护进程、未修改的原生可执行文件、相关资源和 `services/core/deploy/codex/Dockerfile` 的上下文。 +该脚本会检查软件包版本,为 Linux amd64 构建 `oac-daemon`,并准备一个仅包含守护进程、未修改的原生可执行文件、相关资源、共用的登录 profile 构建步骤和 `services/core/deploy/codex/Dockerfile` 的上下文。 **Claude Code Runtime 镜像。** 必须使用 Node 20 或更高版本以及 pnpm。 @@ -103,16 +105,16 @@ make build-e2b-provider Docker 使用固定版本的 CPython 和 Debian 12 镜像按 `GOARCH=amd64`(默认)或 `GOARCH=arm64` 构建 Linux 辅助程序。Python 依赖闭包(including PyInstaller)在 `services/core/tools/e2b-provider/requirements.lock` 中按哈希锁定;不需要 E2B 账户密钥。要使用其他输出目录,请设置 `E2B_PROVIDER_BUILD_DIR`。构建结果完全由辅助程序源代码、`LICENSE` 和构建脚本决定,因此会按它们的哈希缓存在 `~/.oac/cache/e2b-provider/` 下,仅在它们变化时重新构建。输出为 `oac-e2b-provider-linux-.tar.gz` 及其 `.sha256`;解压后会得到 `oac-e2b-provider/`,其中包含可执行文件、`_internal/`、`licenses/`、`requirements.lock` 和 `manifest.json`。Core 镜像使用该目录树;主机需要兼容的 glibc 和 CA 证书,而不需要 Python。 -**microsandbox 辅助程序。** 仅支持 Linux,并且需要 C 编译器: +**microsandbox 辅助程序。** 仅支持 Linux amd64,并且需要 C 编译器: ```sh make build-microsandbox-provider make check-microsandbox-provider ``` -该辅助程序会写入 `~/.oac/build/microsandbox-provider/oac-microsandbox-provider`。其独立的 Go 模块固定 microsandbox Go SDK v0.7.2,并嵌入匹配的 FFI 库;构建生产版本时,绝不能使用该 SDK 的 `microsandbox_ffi_path` 标签。Core 本身仍采用禁用 CGO 的构建。该辅助程序需要 glibc,并且只能在节点上运行。 +该辅助程序会写入 `~/.oac/build/microsandbox-provider/oac-microsandbox-provider`。其独立的 Go 模块固定官方 microsandbox SDK 源码提交。上述两个命令与分发构建均使用 `scripts/build-microsandbox-provider.py`:它暂存模块依赖,验证匹配的官方 FFI 发行文件校验和,填充 SDK 的空发行包,并构建嵌入该 FFI 的程序。SDK 源码保持不变,仓库不提交 vendored 二进制文件。请使用此入口,而非直接调用 `go build`;构建生产版本时,绝不能使用该 SDK 的 `microsandbox_ffi_path` 标签。Core 本身仍采用禁用 CGO 的构建。该辅助程序需要 glibc,并且只能在节点上运行。 -**microsandbox 运行时。** 分发包使用官方的 [v0.7.2 release](https://github.com/superradcompany/microsandbox/releases/tag/v0.7.2) 归档 `microsandbox-linux-x86_64.tar.gz`,SHA256 为 `47c223e3ef5298abf05f47ed9f87981106e400d99bb3f1d042d4d6881346b18b`(即 `scripts/core-distribution-manifest.py` 中的 `RUNTIME_ARCHIVE_SHA256`)。构建过程会先验证校验和,再解压 `msb` 和 `libkrunfw.so.5.6.1`,并记录这两个文件的哈希。辅助程序会在每次调用时检查这些哈希,并且绝不安装或升级它们。 +**microsandbox 运行时。** 分发包使用官方的 [v0.7.8 release](https://github.com/superradcompany/microsandbox/releases/tag/v0.7.8) 归档 `microsandbox-linux-x86_64.tar.gz`,SHA256 为 `86f9f72dc3e639c7175bc07909b4b63ce412517c1ef8a2e1921171af5682fded`(即 `scripts/core-distribution-manifest.py` 中的 `RUNTIME_ARCHIVE_SHA256`)。构建过程会先验证校验和,再解压 `msb` 和 `libkrunfw.so.5.6.1`,并记录这两个文件的哈希。辅助程序会在每次调用时检查这些哈希,并且绝不安装或升级它们。 ### 独立 Core 构建 {#standalone-core-builds} diff --git a/docs/zh/runtime-protocol.md b/docs/zh/runtime-protocol.md index 8d6a2975c..6cb249e5d 100644 --- a/docs/zh/runtime-protocol.md +++ b/docs/zh/runtime-protocol.md @@ -1,7 +1,7 @@ --- title: "Core–Runtime 协议" source: docs/runtime-protocol.md -source_hash: 7c99f40b6c714885cbe279d005062bf631e2e145abba2b8298533aa31157df27 +source_hash: a3a06d0f5db73286e87f8784df45e304cb18685c43ede7c90880b127cacc453c --- 此协议在 Runtime daemon 获取机器凭据后连接 Core 与 daemon,定义 daemon 连接上消息的含义和顺序。wire 类型、限制和验证器仅在 [`internal/agentdaemon/proto`](https://github.com/MiniMax-AI/OpenAgentCore/tree/main/internal/agentdaemon/proto) 中定义一次;Core 的 [gateway](https://github.com/MiniMax-AI/OpenAgentCore/tree/main/services/core/internal/runtimegateway) 与参考 Runtime 的 [dispatcher](https://github.com/MiniMax-AI/OpenAgentCore/tree/main/apps/daemon/internal/dispatch) 都使用它们,因此无需同步第二套 payload schema。签发凭据和打开连接的 HTTP 路由见[机器连接 API](../../contracts/agents-api/zh/machine-api.md)。 @@ -45,6 +45,7 @@ wire 上每个字段都是 JSON boolean,所有字段都必须出现,包括 ` | `local_environment`, `workspace_read_preparation`, `workspace_output_export` | Environment 类型为 `openai_hosted` 或 `self_hosted` | | `workspace_read_preparation` | 空闲 Files 目录读取需要只读 preparation | | `native_session_recovery` | Session 已启动过 Turn,但未记录原生 Session ID | +| `retained_native_history` | Environment 使用外部工作区绑定;自带工作区存储不要求此能力 | | `web_search_control`, `text_verbosity` | Harness 的 engine profile 声明该控制 | | `structured_output` 和 `message_items` | Agent 请求 `json_schema` 输出 | | `subagent_observations` | `multi_agent.enabled` 为 true | @@ -55,6 +56,8 @@ wire 上每个字段都是 JSON boolean,所有字段都必须出现,包括 ` | `message_images`, `function_result_images` | 消息或 function result 携带图像 | | `mcp_http_tools`, `mcp_http_required`, `mcp_http_bearer_auth` | Agent 声明 HTTP MCP server;其中一个为 `required`;其中一个选用了 Vault 凭据 | +`retained_native_history` 表示 Runtime 与 Harness 组合可在确认原生进程关闭后,从私有持久状态重新打开同一个原生 Session。它不保证恢复进程内存、后台进程或外部连接。所选文件系统还必须独立提供持久存储。Core 根据旧设备已持久化的声明判断能否释放计算而不终结 Session,并在准入执行前验证新 Runtime。原生历史缺失或无效时明确失败,绝不授权新建原生会话或重放此前输入。 + Core 对 `usage` 和 `resume` 没有准入规则。 `execution_prepare` 的配置携带 Core 为各 Run 设置的显式启用项: @@ -111,7 +114,7 @@ Usage frame 和最终 usage snapshot 都携带当前执行的累计测量,替 ## 准备与执行顺序 {#preparation-and-execution-order} -无论托管还是用户自有环境,每条连接都通过 `runtime_prepare` 初始化 Environment;[Environment 契约](../../contracts/agents-api/zh/environments.md#runtime-capability-preparation)负责准备内容和时机。传输文件或 archive 时,发送 `begin`,等待 `ready`,发送有序 chunk 并等待每个匹配的 `received` offset,再发送 `commit` 并等待 `completed`。初始化和终结阶段使用不含文件数据的类型化 header。使用共享 validator 验证预期结果、offset、size 和有限错误 code。每条连接允许一个 transfer。chunk 回执确认暂存字节,不确认安装;完成的 commit 确认该操作,不证明后续 Turn 已运行。 +无论托管还是用户自有环境,每条连接都通过 `runtime_prepare` 初始化 Environment;[Environment 契约](../../contracts/agents-api/zh/environments.md#runtime-capability-preparation)负责准备内容和时机。传输文件或 archive 时,发送 `begin`,等待 `ready`,发送有序 chunk 并等待每个匹配的 `received` offset,再发送 `commit` 并等待 `completed`。初始化和终结阶段使用不含文件数据的类型化 header。每个 `begin` 必须携带 `budget_ms`,取值为 1 至 1800000 的整数,表示 Core 所拥有的初始化预算剩余毫秒数;chunk 和 commit 不携带该字段。Runtime 从收到 `begin` 起执行此相对预算,Core 的等待不超过原操作期限。完整的 `begin`–`commit` 暂存阶段另有两分钟限制;应用已提交操作及等待 `completed` 使用剩余初始化预算,不受暂存期限限制。应用前到期会拒绝传输;应用中到期则在本地变更停止后报告 `unknown`,绝不授权重放。使用共享 validator 验证预期结果、offset、size 和有限错误 code。每条连接允许一个 transfer。chunk 回执确认暂存字节,不确认安装;完成的 commit 确认该操作,不证明后续 Turn 已运行。 执行 Turn 分为五步: @@ -125,7 +128,10 @@ preparation 预约每个 Turn 的准入,而不是新 Executor。它携带明 preparation 和 start 在 receive loop 与 router lock 之外运行。admission 在授予五分钟后到期,重试不延长截止时间;到期不解除 Runtime 完成清理结算的义务。Runtime 分别限制活动 preparation、execution 和保留的空闲资源,关闭中或不确定资源持续计入限制,直到清理成功。明确的 `execution_prepare` 拒绝若为 `preparation_capacity`,会让排队 Turn 保持未领取,供 Worker 重试,包括清理占用容量的情况;其他错误或不确定交付都不授权重放。Runtime 最多保留 64 条 admission 记录,旧 handle 不会消耗替代项的 admission。这些记录仅属于连接,不是持久化输入重放。 -Executor 空闲到期属于 Runtime 资源策略,与 Core 的活动 Turn 并发限制独立。关闭时 Runtime 关闭活动和空闲 Executor,保留关闭失败的目标,并允许稍后串行重试。普通断连会关闭失败的 transport 并保留原 router,直到 shutdown 成功;等待超时或清理失败不授权重连,进程 shutdown 继续等待,不丢弃自己拥有的原生资源。工作区操作在跨 Turn 和 Executor 关闭后仍保留绑定与结算规则。 +Executor 空闲到期属于 Runtime 资源策略,与 Core 的活动 Turn 并发限制独立。关闭时 Runtime 关闭活动和空闲 Executor,保留关闭失败的目标,并允许稍后串行重试。普通断连会关闭失败的 transport 并保留原 router,直到 shutdown 成功;等待超时或清理失败不授权重连,进程 shutdown 继续等待,不丢弃自己拥有的原生资源。对于 Runtime 初始化,shutdown 会等待同步 apply 操作和回执发送结束后才释放连接;`unknown` 结果仍是不确定结果,并阻止该 Router 上的后续准备,但本地变更停止后不再阻止重连。Core 保留失败的初始化状态,绝不重放初始化。工作区操作在跨 Turn 和 Executor 关闭后仍保留绑定与结算规则。 + + +挂起先关闭准入、排空已接纳工作与回执,同时保留已结算的空闲 Executor,随后才确认 `environment_quiesced`。活动、准备中、失效或其他未结算的所有者会阻止挂起。整个挂起期间停止空闲到期计时并隔离其回调;精确匹配且已认证的 `environment_resume` 为同一批所有者恢复一个新的空闲计时间隔。只有这种计划性重连跨 socket 保留 Router 与原生所有者;普通断连、shutdown 和连接生命周期取消仍要求确认清理完成。排空失败后,准入保持关闭直到 shutdown。恢复后的准备仍检查相同的原生身份与不可变配置,包括凭据冲突;绝不静默替换配置冲突的驻留 Executor。Sandbox Provider 仍负责其 checkpoint 实现的文件系统刷新和计算停止保证;Runtime 静止本身既不证明这些保证,也不证明原生 RAM 或外部网络连接已恢复。 ## 活动输入回执 {#active-input-receipts} diff --git a/docs/zh/sandbox-provider.md b/docs/zh/sandbox-provider.md index 3a6eb2e18..462fc10f6 100644 --- a/docs/zh/sandbox-provider.md +++ b/docs/zh/sandbox-provider.md @@ -1,7 +1,7 @@ --- title: "添加 Sandbox Provider" source: docs/sandbox-provider.md -source_hash: 5b3dd49c6f539c265a07d0ab44779b27192f4de378b41a0e946487faef6ab5b3 +source_hash: fc14797b2af53ed0e2bd64828c83c19e711e5bf30bb1cdfa24574be522b1c13c --- **Sandbox Provider** 为 Core 管理的 Environment 提供 Runtime daemon 运行所需的外层计算资源,以及启动 daemon 的有界引导流程。本指南说明如何添加 Provider,并作为 Core 驱动 Provider 的参考。接口为 [`SandboxProvider`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/services/core/internal/sandbox/sandbox_provider.go)。 @@ -168,7 +168,7 @@ allocation、专用 daemon credential digest 和精确 Session binding 在 `Crea 配置 provider 后,Worker 扫描已提交且没有 allocation 的 pending hosted Environment,涵盖空闲 Session 创建以及 commit 与 bootstrap 之间中断后的恢复;已有 allocation 不重新进入此路径。scan 有界,由 lifecycle owner 串行化,不需要调用方操作。没有 Turn 的初始预约让 Session 保持空闲,daemon 连接不被当作原生 readiness。同一 scan 在验证精确 Session、device binding 和已结算 bootstrap 后,发布带持久 generation 的认证连接观测。 -Core 在 Turn 之间检查已连接且已观察的计算资源仍是其 Session 正在运行的 allocation;该检查不做任何修改,也不复活 cleanup 请求。正在运行的计算资源不会到期:显式删除和保留状态期限授权其清理。停止或缺失 container 不授权丢弃保留工作区或历史。禁用 provider 停止新 hosted admission 与 bootstrap,但不阻止现有 Session 的取消、function result 或 input retry outcome。 +Core 在 Turn 之间检查已连接且已观察的计算资源仍是其 Session 正在运行的 allocation;该检查不做任何修改,也不复活 cleanup 请求。正在运行的计算资源不会到期:显式生命周期清理和保留状态期限策略管理资源回收。停止或缺失 container 不授权丢弃保留工作区或历史,也不授权选择替代计算资源。禁用 provider 停止新 hosted admission 与 bootstrap,但不阻止现有 Session 的取消、function result 或 input retry outcome。 终结清理原子撤销 device authority、记录 Environment 失败或到期、结算 pending input 并请求取消,然后才调用 `Kill`;原 input deadline 与 retry outcome 保留。临时 provider outage、未知 Create result 和停止的计算资源不证明永久失败。公开 Session 删除后 Core 保留 allocation,仅在所属 compute 与 volume 清理完成且原 Create 已结算的证明成立后标记 released;未知创建即使观察到不存在也保留 cleanup ownership,有界 scan 继续捕捉延迟资源,不再调用 `Create`。 @@ -178,38 +178,64 @@ Core 在 Turn 之间检查已连接且已观察的计算资源仍是其 Session 每个注册 node 有一个串行 lifecycle worker,负责 gate、allocation 与 pending cursor、connection 和 wake hint;E2B allocation 共享一个没有 node 的串行 lifecycle。薄 coordinator 发现 node 并关闭 worker,数据库、provider 或等待操作期间不持有 map mutex。worker 独立推进,因此一个在线 node 的 provider 卡住不会阻塞其他 node:lifecycle 并发为每 node 一项操作,随 node 数量增长。离线 worker 保留,因此其保留资源在重连后仍可观察。 -allocation scan 在应用 32 行分页限制前按 node 过滤,pending scan join 尚未释放的已提交 placement。每个 node 推进自己的 cursor,包括越过失败观测,并在末尾回绕一次。direct provisioning 在进入该 node gate 前解析 tenant 范围内 placement,已有 allocation 必须与其一致;Core 不选择另一 node。 +allocation scan 在应用 32 行分页限制前按 node 过滤,pending scan join 尚未释放的已提交 placement。每个 node 推进自己的 cursor,包括越过失败观测,并在末尾回绕一次。direct provisioning 在进入该 node gate 前解析 tenant 范围内 placement,活动 allocation 必须与其一致。只有确认释放后才能建立新的 placement;断连不会迁移活动 allocation。 释放 execution lease 前,coordinator 停止接受工作,取消并排空每个 node worker 和 direct caller。lease 丢失影响全部;普通 provider failure 限于所属 node。计划 deployment drain 或 node retirement 通过五秒有界 lease gate,在 leased operation 之间同步取消 lifecycle context,包括活动 manual reconcile,不仅为了改变配置就取消进行中的 leased query。失败的 cancellation fence 关闭 manager admission 并报告 owner failure。失败的 retirement 保留原 lifecycle identity 和 gate,直到 owner shutdown,drain barrier 保持关闭。Session lock、deployment capacity transaction 和 revision-checked receipt 仍是权威依据,外部操作不持有数据库锁。 ### Placement 与容量 {#placement-and-capacity} -placement 自动完成:environment-to-node placement 与 Session 创建及其 retry identity 一起提交,调用方不能选择 node,重试即使 node 离线也保留原 node。Node capacity 计入 pending reservation 和未决资源,新 placement 与 suspended-to-restoring 转移共享数据库锁。未知操作保留预约,source teardown 必须确认后才能释放 active capacity,确认 cleanup 后释放 placement capacity。保留所有权需要精确 provider evidence:socket path、缺失 instance 或空列表都不证明 cleanup,也不授权 replacement。 +placement 自动完成:Session 创建提交持久的 pending 工作,共同调度器在 allocation 之前预留兼容节点。[部署契约](../../contracts/agents-api/zh/sandbox-deployment.md#generation-ownership-and-rollout) 定义等待、排序与代次选择。调用方不能选择 node;活动 allocation 即使在 node 离线时也保留原 node。确认 allocation 释放后,符合条件的保留 Session 可以重新预约兼容容量,包括其他 node。Node capacity 计入 pending reservation 和未决资源,新 placement 与 suspended-to-restoring 转移共享数据库锁。未知操作保留预约,source teardown 必须确认后才能释放 active capacity,确认 cleanup 后释放 placement capacity。保留所有权需要精确 provider evidence:socket path、缺失 instance 或空列表都不证明 cleanup,也不授权 replacement。 -部署的 CPU、memory、disk 设置、`max_active`、`max_retained` 和保留状态期限限制每个 node。不提供 node-level drain、跨 node Session migration、multi-active Core、autoscaling 或 snapshot replication。node 持有 allocation、保留状态、reservation、unknown result 或 cleanup 时拒绝移除 node,离线 ownership 保留。 +部署的 CPU、memory、disk 设置、`max_active`、`max_retained` 和 retained-state retention 限制每个 node。确认释放后的冷替换不提供 node-level drain、不可达写入方的故障转移、并发写入方、multi-active Core、autoscaling 或 snapshot replication。node 持有 allocation、保留状态、reservation、unknown result 或 cleanup 时拒绝移除 node,离线 ownership 保留。 ### 暂停 {#suspension} 单一 `SandboxProvider` 一并声明 `Initial`、`NewCompute`、`GetCompute`、`RenewCompute`、`Suspend`、`Resume`、`KillCompute`、`DeleteRetained`、`RunCommandCompute` 和 `ResumeCompute`:十项全部支持或全部不支持。生命周期不通过可选暂停接口或 Provider 名称分支选择。 -`RetainedState` 是 adapter 拥有的不透明信封,包含 `Reference`、`ID`、`Data`、`OperationID`、`SourceGeneration`、`SourceName` 和 `SourceID`。`Data` 必须非空且不超过 64 KiB。Core 原样保留信封并校验操作与源实例归属;不可变的 allocation Provider/代次绑定决定唯一有权解释它的 adapter。保留状态不必是独立快照:microsandbox 将完整原生快照身份放入 `Data`,E2B 保留带版本的原生暂停收据,不承诺独立磁盘镜像。adapter 在任何副作用前校验原生内容及归属。 +`RetainedState` 是 adapter 拥有的不透明信封,包含 `Reference`、`ID`、`Data`、`OperationID`、`SourceGeneration`、`SourceName`、`SourceID` 和可选的 `Compatibility`。`Data` 必须非空且不超过 64 KiB。Core 原样保留信封并校验操作与源实例归属;不可变的 allocation Provider/代次绑定决定唯一有权解释它的 adapter。保留状态不必是独立快照:microsandbox 将完整原生快照身份放入 `Data`,E2B 保留带版本的原生暂停收据,不承诺独立磁盘镜像。adapter 在任何副作用前校验原生内容及归属。 -`Compute.RestoredFrom`、`ComputeState.Retained`、`ResourcesReleased`、`SuspendSettled` 和恢复的 `ReconcileOnly` 是结算证据。资源缺失不证明未知暂停已结束。`RenewCompute` 只续租精确实例;不确定结果不授权第二次创建、捕获或恢复。持久 `runtime_compute` 使用 `protocol_version: "1"`;启动拒绝缺失或未知版本。现有版本 1 信封可直接读取,无需改写。不会隐式转换只含 snapshot 的形状,也不会合成重放或丢弃保留状态。 +`Compute.RestoredFrom`、`ComputeState.Retained`、`ResourcesReleased`、`SuspendSettled` 和恢复的 `ReconcileOnly` 是结算证据。资源缺失不证明未知暂停已结束。`RenewCompute` 只续租精确实例;不确定结果不授权第二次创建、捕获或恢复。持久 `runtime_compute` 使用 `protocol_version: "1"`;启动拒绝缺失或未知版本。仅匹配当前 retained-state 字段形状的版本 1 信封可直接读取,无需改写。旧 snapshot-only 形状即使标记相同的版本 1 也会被拒绝;数字相同不证明互通。升级不做迁移、合成重放或丢弃保留状态。 `ResumeRequest.Workspace` 携带 allocation 独立拥有的文件系统绑定;Core 在恢复前获取 ready 绑定。源实例必须已释放活跃执行能力,目标才可成为写者。adapter 在原生副作用前校验绑定;不支持独立文件系统的 adapter 拒绝非空绑定。`DeleteRetained` 仅删除 adapter 的计算保留状态,绝不删除 workspace。显式 Session 删除先完成 compute 清理,再删除独立存储,恢复期间保持单一活跃写者。 -Core 对所有声明暂停支持的 Provider 应用统一固定策略:工作空闲 5 分钟(300 秒)后暂停,保留状态期限为 24 小时(86400 秒);部署的 [`suspension`](../../contracts/agents-api/zh/sandbox-deployment.md#safe-response) 返回这些值。Core 暂停已经初始化的 Environment,包括尚未执行 Turn 的 Environment,前提是没有 root 或 Subagent Turn 排队、执行或等待,没有输入、文件操作或初始化待处理,且真实活动已空闲达到该时长。Core 在同一事务中使用数据库时钟记录 allocation 的初始化完成和 root 或 child 终结转换。候选筛选和持有 Session 锁的复查比较数据库经过时间与空闲时长,初始保留状态期限也锚定同一数据库观测,因此 Core 与数据库主机的时钟无需一致。公开历史中的原生完成时间戳保持不变,但不驱动空闲准入,心跳也不重置活动时间。确认计划暂停前,daemon 关闭准入并排空原生清理、输出收据和文件工作。 +Core 对所有声明暂停支持的 Provider 应用统一共享策略:通常在工作空闲 5 分钟(300 秒)后暂停,保留状态期限为 24 小时(86400 秒);部署的 [`suspension`](../../contracts/agents-api/zh/sandbox-deployment.md#safe-response) 返回这些值。Core 暂停已经初始化的 Environment,包括尚未执行 Turn 的 Environment,前提是没有 root 或 Subagent Turn 排队、执行或等待,没有输入、文件操作或初始化待处理,且真实活动已空闲达到该时长。空闲时钟不早于 allocation 进入 running compute phase 的时刻,包括唤醒后。Core 在同一事务中使用数据库时钟记录 allocation 的初始化完成和 root 或 child 终结转换。候选筛选和持有 Session 锁的复查比较数据库经过时间与空闲时长,初始保留状态期限也锚定同一数据库观测,因此 Core 与数据库主机的时钟无需一致。公开历史中的原生完成时间戳保持不变,但不驱动空闲准入,心跳也不重置活动时间。确认计划暂停前,daemon 关闭准入并排空原生清理、输出收据和文件工作。 + +容量压力可以让已经空闲的 node allocation 在通常超时之前暂停,但仍需经过 15 秒无活动宽限期。Session 锁内的决策重新检查 pending work、wake request 和 activity,再获取共享 deployment 容量锁并确认兼容的等待需求。它不会中断活动 Turn,也不会在普通 quiesce、capture 和 source-stop 证明完成前释放容量。压力触发的回收采用保守策略:只有进行中回收所属的节点当前有资格服务同一需求时,它才会推迟另一次回收。不可用或不兼容的节点保留其 ownership receipt,但不会阻挡健康节点的容量。如果暂停无法使容量可用,等待请求仍保留原始期限。 Worker 租约、Session 锁和每个 node 的 gate 负责所有 Provider 的暂停。新 Turn claim、文件写入意图和捕获准入在 Session 锁下串行化,共享计算阶段检查;新待处理工作取消捕获并唤醒同一源实例。正常准备在经过认证的恢复握手后等待计算阶段变为 running;待处理输入的提升与生命周期转换冲突时,输入保持 pending。计算阶段和经 revision 校验的收据存储在 allocation 中。Core 在副作用前持久化静止、捕获和恢复意图,仅新收据执行捕获或恢复;恢复流程观察精确尝试,不重试结果未知的创建、捕获或恢复。已消费的保留状态不会使运行中的代次回滚。删除、撤销和保留期到期始终优先于唤醒,直至最后的数据库 compare-and-swap;未知清理身份会一直保留,直到确认所属资源已不存在。已消费产物和旧计算实例会被删除,因此暂停循环不会累积可写磁盘链。 -排队工作和实时 Environment 文件访问会唤醒暂停的 Environment;历史和已发布 Artifact 的读取不会唤醒。计划暂停在 daemon 连接上使用 Environment 与 suspension token。由 PID 和启动时间隔离的本地控制信号(`RunCommandCompute`)唤醒 parked daemon,daemon 在准入工作前重新认证。确认前的临时断连通过有界尝试和退避重试同一已准备好的暂停;永久认证或协议拒绝会将其关闭。Core 负责保留状态的到期期限,daemon 没有相应定时器。静止确认丢失时可以通过明确回滚解冻同一源实例,但不授权捕获。 +排队工作和实时 Environment 文件访问会唤醒暂停的 Environment;历史和已发布 Artifact 的读取不会唤醒。实时文件请求在进入文件工作队列前,等待当前 Runtime 通过 credential 授权的连接及 Harness 声明;compute Create 完成并不代表 Runtime 已就绪。direct 保留状态通过 allocation 对应的 adapter 恢复;node 检查点遵循[检查点转移](#checkpoint-transfer)中的兼容性与 placement 规则。计划暂停在 daemon 连接上使用 Environment 与 suspension token。由 PID 和启动时间隔离的本地控制信号(`RunCommandCompute`)唤醒 parked daemon,daemon 在准入工作前重新认证。确认前的临时断连通过有界尝试和退避重试同一已准备好的暂停;永久认证或协议拒绝会将其关闭。Core 负责保留状态的到期期限,daemon 没有相应定时器。静止确认丢失时可以通过明确回滚解冻同一源实例,但不授权捕获。 -`Suspend` 负责原生资源释放,返回绑定的保留句柄、suspended 状态、`ResourcesReleased` 和 `SuspendSettled` 后,Core 才释放活跃容量。`ReconcileOnly` 禁止重放原始捕获或暂停,但允许完成由持久保留产物证明安全的 adapter 清理。无保留状态的结果只有在带有 `SuspendSettled` 且源实例处于可恢复的运行或暂停状态时才允许回滚。其他所有不确定结果均保留所有权并关闭准入。Core 从不在 `Suspend` 后无条件销毁源实例。 +`Suspend` 负责原生资源释放,返回绑定的保留句柄、suspended 状态、`ResourcesReleased` 和 `SuspendSettled` 后,Core 才释放活跃容量。`ReconcileOnly` 禁止重放原始捕获或暂停,但允许完成由持久保留产物证明安全的 adapter 清理。无保留状态的结果只有在带有 `SuspendSettled` 且源实例处于可恢复的运行或暂停状态时才允许回滚。其他所有不确定结果均保留所有权并关闭准入。Core 从不在 `Suspend` 后无条件销毁源实例。 没有保留状态时,`SuspendSettled` 还证明已持久关闭该精确源实例与操作的所有迟到原生派发。进程内调用栅栏、分配锁或当前原生资源不存在,都不能单独证明这一点。E2B 与 microsandbox 均维护有界的 adapter 私有准入日志;回滚不能清除同代次关闭记录,只有已验证的更高源代次可以推进栅栏。 `Resume` 将保留状态恰好消费一次并恢复到预先提交的目标。恢复逻辑观察同一次尝试。Core 持久化 `waking`、认证并恢复 daemon、删除已消费的保留资源,然后提交 `running` 并准入工作。`DeleteRetained` 是幂等产物清理,会保留运行中的计算资源。清理失败会保持 `waking` 阶段,不能触发再次恢复。`KillCompute` 仍然是破坏性操作;清理旧代次时不得终止共享原生 ID 的较新活跃实例。 现有 Session 锁、生命周期租约、空闲规则、容量查询和清理顺序继续作为权威。每个尚未释放的分配都占用 `max_retained`,包括运行中的分配。每个分配在预留时都写入共享计算协议版本,包括暂停阶段为 `disabled` 的分配。激活会拒绝任何协议版本缺失或不同的未释放分配;升级前必须由旧版本完成普通清理。Session 历史保留。 +### 协调协议升级 {#coordinated-protocol-upgrade} + +Core 与 node 一同升级到 node wire version 8;E2B helper 使用 private wire version 4,并重建包含托管暂停控制的模板;microsandbox helper 使用 private wire version 5。所有 Provider 交付 Runtime bootstrap version 2。每个边界独立验证自身契约,不同边界的版本号不能互换。激活前,使用此前兼容的版本对不兼容的未释放 allocation 完成正常清理。启动栅栏保留其持久收据及 Session 历史,不通过 Provider 调用迁移它们。 + +### 检查点转移 {#checkpoint-transfer} + +支持检查点的 generation 报告 `CheckpointCompatibility`,包含不透明的 `ArtifactDomain` 与 `ExecutionClass` token。可转移的 `RetainedState` 在 `Compatibility` 中携带两个 token。direct 保留状态可省略它们,但省略不提供跨 node 恢复或清理资格。E2B 原生暂停不声明检查点可迁移性。Core 只比较这些值,不解释 CPU 特性、文件系统路径或存储实现。恢复目标必须在线、对 allocation 的不可变 deployment generation 已就绪、两个 token 均匹配,并有 active 容量;跨 node 时还需 retained slot。allocation、Device 和 Session 身份保持不变。Session 与 deployment 事务先提交目标路由、placement 和精确 restore intent,再执行目标侧原生操作。没有容量或兼容目标时,保留快照的所有权一直持续到配置期限。目标必须已持有该精确 generation;Core 不会在新节点自动准备历史 generation。升级期间应保留符合条件的节点及其 generation provider,直到检查点保留义务结束。 + +对于可转移检查点,`Suspend` 只有在持久发布已验证的完整归档、停止精确 source compute,并结清 source 本地 capture 与 cleanup 义务后,才能报告 `ResourcesReleased` 和 `SuspendSettled`。Core 先持久化绑定的保留状态,再标记 suspended;仅原生资源不存在并不足够。`Suspend.ReconcileOnly` 可以为同一 operation 已验证的快照完成归档发布和源清理,但不能重新捕获快照,也不能停止已经恢复 running 的源。此前已发布的归档缺失或损坏时,`Suspend.ReconcileOnly` 可以返回其精确持久 ownership receipt,标记 `Status: unknown` 和 `ResourcesReleased: false`,供清理使用;该 receipt 不证明归档当前可用,也不证明源不存在。`Resume` 仍在执行前验证归档。完成源结清屏障后,source node 不可用不再阻止恢复。删除和正常到期可以将空闲 suspended allocation 的清理路由转给同一 artifact domain 内的在线 node;删除不要求 execution class 匹配。清理目标必须已提供同一不可变 generation,且有 retained 容量。没有这样的 node 时,清理保持 pending,不释放所有权;运维需恢复符合条件的 node 或其容量。未知 running 或 restoring writer 不会仅因 node 不可达而迁移。 + +恢复用 `Resume.ReconcileOnly` 观察已持久化的 target 与 operation。`RestoreAttemptClosed` 是精确 operation ID,不是通用 absence 标志:adapter 只有在持久关闭从未进入原生执行的 attempt,并隔离该 ID 的所有迟到请求后才能返回。如果 adapter 能证明当前调用在派发原生恢复之前失败,也可以关闭已持久 admitted 的 attempt。microsandbox adapter 仅在 fresh 请求的 wire deadline 于原生派发之前到期时执行这种关闭;归档损坏、资格错误和不透明 SDK 失败都不授权自动重试。Core 随后先持久化新的 operation ID,再尝试执行。一旦已派发原生恢复,结果未知的 attempt 保留相同 target、所有权与原始 retention deadline;原生列表缺失、helper 死亡或超时都不授权重放。这种恢复可能一直不可用,直到显式删除或正常保留期到期;等待中的 input 可能达到其公共 deadline。用户删除仍需精确原生 settlement,之后才能释放资源与所有权。 原地恢复的 adapter 可以保留非空 native ID:关闭证明必须与请求的 ID、逻辑代次、名称和保留状态来源完全一致。该证明说明本次恢复未派发,不表示底层已暂停源实例不存在。 + +microsandbox adapter 要求显式的私有 `checkpoint_root`,位于所有 guest 可访问文件系统之外。目录权限为 `0700`,私有 namespace marker 标识真实归档存储。默认安装创建 node 本地私有 store。跨 node 恢复要求运维为参与 node 挂载同一个私有 store,挂载路径可以不同。Runtime state 目录与 workspace binding 都不用于推导该 root。adapter 在恢复前再次验证归档、外部文件系统身份和原生 execution class。恢复 RAM 与文件描述符不保证外部 TCP 连接连续,也不保证应用副作用可安全重放。 + +### 检查点保留期后的冷替换 {#cold-replacement-after-checkpoint-retention} + +保留状态期限限制计算资源,不限制符合条件的保留 Session 的生命周期。到期后,只有独立文件系统对象已就绪、初始化已完成、Session 与 Environment 均未终结,且此前已认证 Runtime 的 `retained_native_history` 声明已持久化时,Core 才能在不终结 Session 的情况下清理 allocation。能力及原生恢复义务由 [Core–Runtime 协议](runtime-protocol.md)定义。仅保留文件系统并不足够。 + +Core 在释放 allocation 和 placement 归属前,确认原始创建已结算且所属计算资源与保留状态均已清理。替代计算资源准入前撤销旧 device 的权限。Create、Kill 或 DeleteRetained 的结果尚未确认时,Core 保留归属,不准入替代写入方。可恢复的 Session 与 Environment 保持断连,保留同一文件系统对象和初始化结果。归档、重置和删除优先,唤醒请求不能撤销它们。 + +容量压力不会缩短检查点保留期。在配置的期限到达前,已挂起的 allocation 保留快照并继续占用 retained slot;容量不足或节点不兼容不允许降级为冷替换。等待请求保持原有的输入期限。保留期到期后,常规 expiry cleanup 必须确认资源已删除,其他 allocation 才能预留该 slot;文件与已验证的原生历史继续保留。 + +新 input 或实时 Environment 文件访问会在兼容 node 上预约新 allocation,并引导新的 Runtime device。准备流程复用同一文件系统和已完成的初始化,不重放初始化。验证替代 Runtime 的声明后,执行从保留历史续接同一原生 Session。原生历史缺失或无效时明确失败;Core 不替换为新的原生 Session,也不重放已完成的 input。读取历史和已发布 Artifact 不预约计算资源。此路径不保留 VM 内存、后台进程或开放连接,也不为未声明能力的 Runtime、Harness 或文件系统组合提供资格。 + ### 重置与归档 {#reset-and-archive} [reset](../../contracts/agents-api/zh/sandbox-deployment.md#reset) 是 runtime manager 在计数工作之外推进的持久 execution state。start、escalation、cancellation、setup、update 和 finalization 通过 mutation gate 串行化。空闲状态先在 Session lock 再在 deployment lock 下复查,finalization 时不反向加锁。工作使用 keyset paging,绑定 reset request time 和 generation,并持久化 absolute deadline 与已验证 audit provenance。 @@ -260,4 +286,4 @@ node 使用 [provider 配置](configuration.md#docker-node-configuration)中的 `DeploymentPolicy.Workspace` 声明挂载要求;未声明表示不支持外部存储。派生的不可变 `DeploymentSpec.Workspace` 回执选择代次的模式和能力;[部署流程](../../contracts/agents-api/zh/sandbox-deployment.md#resources)不会将文件系统配置复制进代次。Microsandbox 要求 `host_directory` 和 `user_xattr`;Docker 和 E2B 拒绝外部挂载。`ValidateWorkspacePolicy` 使用文件系统协议的组合校验器:不强制容量配额的外部文件系统拒绝正值 `EnvironmentDiskMiB`,零表示不请求配额。根磁盘容量仍为必需。未提供外部声明时,继续使用自有磁盘的容量边界。 -Microsandbox 在每次 Create 和 Resume(包括观察中断的恢复)时解析绑定,仅向私有 helper 传递本地目录和不可变 ObjectID。Helper 挂载整个 `/environment`,包含 workspace、staging、initialization 和 packages;私有 HOME 及 Harness 历史仍保存在 VM 根磁盘检查点中。原生资源通过明确的模式、对象和路径标签校验,不根据挂载形状推断模式。恢复校验快照对象,通过 SDK `Volumes` 重映射 `/environment`,要求严格外部挂载策略,并拒绝任何恢复警告。恢复部分失败时保留精确目标身份用于清理,不证明就绪。未知结果继续观察原操作。原生 Bind 检查点及 guest 内 flock 连续性证据不证明跨 VM 隔离。 +Microsandbox 在每次 Create 和 Resume(包括观察中断的恢复)时解析绑定,仅向私有 helper 传递本地目录和不可变 ObjectID。Helper 挂载整个 `/environment`;保留内容与私有原生历史遵循[工作区文件系统范围](workspace-provider.md#lifetime-and-filesystem-scope)和 [Runtime 资源目录](configuration.md#runtime-resource-directories)。私有 HOME、凭据和计算根磁盘不是持久历史存储。原生资源通过明确的模式、对象和路径标签校验,不根据挂载形状推断模式。检查点恢复校验快照对象,通过 SDK `Volumes` 重映射 `/environment`,要求严格外部挂载策略,并拒绝任何恢复警告。恢复部分失败时保留精确目标身份用于清理,不证明就绪。未知结果继续观察原操作。原生 Bind 检查点及 guest 内 flock 连续性不证明对未知 writer 的隔离。 diff --git a/docs/zh/workspace-provider.md b/docs/zh/workspace-provider.md index 489696153..8c681069a 100644 --- a/docs/zh/workspace-provider.md +++ b/docs/zh/workspace-provider.md @@ -1,7 +1,7 @@ --- title: 工作区文件系统 Provider source: docs/workspace-provider.md -source_hash: e9749e1aab1fc2135dcd92a843263243e4adccea393038c495fec67e114dbe83 +source_hash: 21eb220ac59a477a8e799a6de20e01b339536901e8c9685695d2ac498b1b7bdb --- 独立工作区文件系统边界由 [`workspacefs.go`](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/services/core/internal/workspacefs/workspacefs.go) 定义。本文规定必需的集成契约,并不表示所有 Sandbox Provider 或执行位置均已实现。文件系统适配器拥有存储对象与原生挂载解析职责;[Sandbox Provider](sandbox-provider.md) 拥有计算资源职责。Core 在分配任一资源前选择并验证二者的组合。 @@ -43,9 +43,9 @@ source_hash: e9749e1aab1fc2135dcd92a843263243e4adccea393038c495fec67e114dbe83 ## 生命周期与文件系统范围 {#lifetime-and-filesystem-scope} -对象包含整个 `/environment` 目录树,包括工作区、暂存文件、初始化状态和包内容。它们位于同一文件系统,以保留 Environment 准备期间文件系统操作的语义。Harness 的私有原生历史和配置仍属于检查点状态,不重定向到工作区对象。 +对象包含整个 `/environment` 目录树,包括工作区、暂存文件、初始化状态、包内容,以及已具备资格的 Runtime 与 Harness 的私有原生历史。准备数据位于同一文件系统,以保留文件系统操作的语义。原生历史是工作区目录之外的私有状态;位置与设置由 [Runtime 资源目录](configuration.md#runtime-resource-directories)定义。凭据、临时 HOME 和其他计算本地状态不保证持久化。保留原生执行需要 [Core–Runtime 协议](runtime-protocol.md)中的能力声明;仅挂载持久存储并不足以证明支持。 -工作区存储随 Session 保留,直至显式删除 Session。归档、过期、重置和删除计算检查点均不授权删除工作区。Core 仅在显式删除 Session 且确认计算资源停止后调用 `Delete`;未解决的计算或存储变更须保留归属。Session 仍被保留时,创建失败不授权删除其存储。本契约不定义快照过期恢复。 +工作区存储随 Session 保留,直至显式删除 Session。归档、过期、重置和删除计算检查点均不授权删除工作区。Core 仅在显式删除 Session 且确认计算资源停止后调用 `Delete`;未解决的计算或存储变更须保留归属。Session 仍被保留时,创建失败不授权删除其存储。符合条件且未终结的 Session 可以在检查点保留期结束后继续保留,并按[冷替换生命周期](sandbox-provider.md#cold-replacement-after-checkpoint-retention)将同一对象挂载到新的计算资源。此流程不会复活已归档、重置或删除的执行。 ## 内核 NFS 适配器 {#kernel-nfs-adapter} @@ -57,6 +57,10 @@ source_hash: e9749e1aab1fc2135dcd92a843263243e4adccea393038c495fec67e114dbe83 能力声明为 `host_directory`、`user_xattr: true`、`capacity_quota: false`。原生挂载凭据仅包含命名空间标识;解析时根据不可变配置和文件系统所有权验证完整 binding,随后返回根目录下的 `objects//live/data`。仅此数据目录暴露给 guest。租户和应用输入都不提供主机路径。 +NFS 适配器不实施按 Environment 或租户划分的字节或 inode 配额。单个对象可能耗尽共享文件系统,阻止其他对象写入,也可能阻止写入安全删除所需的持久终态元数据。运维人员必须监测可用字节和 inode,并为生命周期元数据和清理保留余量。对整个命名空间设置容量上限,并不提供租户或对象之间的容量隔离。 + 每个对象拥有 `objects//identity`、`staging/`、`live/`、`trash/`,以及删除开始后的 `deleted-marker`。永久标识绑定租户、Environment 和对象 UUID。Create 先准备并同步唯一的私有 staging 封装目录,再原子发布其非空目录为 `live`;重放不会覆盖现有 live 数据。Delete 先持久记录对象本地的终态标记,将 live 数据退役到唯一的对象本地 trash,再先删除数据、后删除其所有权标记。并发删除者收敛。最少量的标识和删除元数据会有意保留;删除不代表所有元数据文件均消失。Create 和 Resolve 拒绝已退役标识。迟到的在途 Create 可能留下空的受控元数据,需要在其系统调用完成后重复 Delete;它不能授权新写入者或复用对象标识。清理只查看该对象的 staging 和 trash,不全量扫描命名空间,也不使用后台服务。 -NFS 存储不提供计算 fencing 或自动分布式故障转移。Guest `flock` 不是跨 VM 写入者锁。上述单写入者和显式删除要求仍然必须遵守。共享文件系统保留 `/environment`,Harness 的私有原生历史仍在计算检查点内;因此,即使工作区文件仍然存在,原生快照过期仍可能导致私有历史丢失。保留工作区不代表原生快照 TTL 到期后仍可恢复原生执行。 +NFS 存储不提供计算 fencing 或自动分布式故障转移。Guest `flock` 不是跨 VM 写入者锁。上述单写入者和显式删除要求仍然必须遵守。确认释放后,可以允许另一兼容 node 上的新 VM 挂载同一对象,但 node 离线或 Create、Kill、快照清理结果未知均不证明旧写入方已停止。原生历史和数据库行为需要对所选 Runtime、Harness、文件系统组合进行资格验证;仅保留文件不保证原生 Session 恢复、跨 node 内存恢复或任意崩溃后的持久性。 + +所选部署代次使用外部工作区存储时,Core 要求所选 Harness profile 支持 `retained_native_history`。创建流程在持有部署锁时校验该代次,然后才提交 Session 或计算预留。既有 Session 的 Runtime 准入和最终执行检查以不可变工作区绑定为依据。自带工作区存储不要求此能力。 diff --git a/internal/agentdaemon/proto/envelope_test.go b/internal/agentdaemon/proto/envelope_test.go index d485d8f7d..cec6e5d6d 100644 --- a/internal/agentdaemon/proto/envelope_test.go +++ b/internal/agentdaemon/proto/envelope_test.go @@ -88,6 +88,7 @@ func TestVersionCompatible(t *testing.T) { ok bool }{ {Version, true}, // exact match + {"0.15.0", false}, // closes idle Executors before suspension {"0.8.99", false}, // patch drift NOT OK {"0.6.99", false}, // minor drift NOT OK {"1.0.0", false}, // major drift NOT OK diff --git a/internal/agentdaemon/proto/inbound.go b/internal/agentdaemon/proto/inbound.go index ed236518f..dfbc36f2d 100644 --- a/internal/agentdaemon/proto/inbound.go +++ b/internal/agentdaemon/proto/inbound.go @@ -158,6 +158,10 @@ type AgentKindCapabilities struct { Usage CapabilitySupport `json:"usage"` Resume CapabilitySupport `json:"resume"` NativeSessionRecovery CapabilitySupport `json:"native_session_recovery"` + // RetainedNativeHistory resumes the same native Session after confirmed + // process shutdown using its retained private state, without replaying input. + // It does not promise process memory or external connection restoration. + RetainedNativeHistory CapabilitySupport `json:"retained_native_history"` Steering CapabilitySupport `json:"steering"` MessageItems CapabilitySupport `json:"message_items"` diff --git a/internal/agentdaemon/proto/prototest/capabilities.go b/internal/agentdaemon/proto/prototest/capabilities.go index 82afd39b5..69cabdaf0 100644 --- a/internal/agentdaemon/proto/prototest/capabilities.go +++ b/internal/agentdaemon/proto/prototest/capabilities.go @@ -17,6 +17,7 @@ func Capabilities(overrides proto.AgentKindCapabilities) proto.AgentKindCapabili Usage: proto.CapabilityUnsupported, Resume: proto.CapabilityUnsupported, NativeSessionRecovery: proto.CapabilityUnsupported, + RetainedNativeHistory: proto.CapabilityUnsupported, Steering: proto.CapabilityUnsupported, MessageItems: proto.CapabilityUnsupported, ToolObservations: proto.CapabilityUnsupported, diff --git a/internal/agentdaemon/proto/runtime_prepare.go b/internal/agentdaemon/proto/runtime_prepare.go index bbb2549c6..576cd8be4 100644 --- a/internal/agentdaemon/proto/runtime_prepare.go +++ b/internal/agentdaemon/proto/runtime_prepare.go @@ -14,11 +14,13 @@ import ( ) const ( - TypeRuntimePrepare = "runtime_prepare" - TypeRuntimePrepareResult = "runtime_prepare_result" - RuntimePrepareMaxBytes = 50 << 20 - RuntimePrepareChunkBytes = WorkspaceWriteChunkBytes - RuntimePrepareMaxFrameBytes = 1 << 20 + TypeRuntimePrepare = "runtime_prepare" + TypeRuntimePrepareResult = "runtime_prepare_result" + RuntimePrepareMaxBudgetMS = 30 * 60 * 1000 + RuntimePrepareTransferBudgetMS = 2 * 60 * 1000 + RuntimePrepareMaxBytes = 50 << 20 + RuntimePrepareChunkBytes = WorkspaceWriteChunkBytes + RuntimePrepareMaxFrameBytes = 1 << 20 ) // RuntimeInitialFile addresses a file within the logical workspace. @@ -38,6 +40,7 @@ type RuntimeInitialization struct { // Runtime resolves logical paths and owns installation destinations. Envelope.ID // identifies one connection-local transfer. type RuntimePreparePayload struct { + BudgetMS int64 `json:"budget_ms,omitempty"` Step string `json:"step"` EnvironmentID string `json:"environment_id,omitempty"` SessionID string `json:"session_id,omitempty"` @@ -64,6 +67,9 @@ type RuntimePrepareResultPayload struct { func ValidRuntimePrepareRequest(p RuntimePreparePayload) bool { if p.Step == "begin" { + if p.BudgetMS <= 0 || p.BudgetMS > RuntimePrepareMaxBudgetMS { + return false + } for _, id := range []string{p.EnvironmentID, p.SessionID} { value, err := uuid.Parse(id) if err != nil || value == uuid.Nil || value.String() != id { @@ -119,7 +125,7 @@ func ValidRuntimePrepareRequest(p RuntimePreparePayload) bool { encoded, err := json.Marshal(p) return err == nil && len(encoded) <= RuntimePrepareMaxFrameBytes } - if p.EnvironmentID != "" || p.SessionID != "" || p.Action != "" || p.Slot != 0 || + if p.BudgetMS != 0 || p.EnvironmentID != "" || p.SessionID != "" || p.Action != "" || p.Slot != 0 || p.Skill != nil || p.Plugin != nil || p.Sources != nil || p.File != nil || p.Initialization != nil || p.SizeBytes != 0 || p.SHA256 != "" { return false } diff --git a/internal/agentdaemon/proto/runtime_prepare_test.go b/internal/agentdaemon/proto/runtime_prepare_test.go index 84519b778..8b63322e9 100644 --- a/internal/agentdaemon/proto/runtime_prepare_test.go +++ b/internal/agentdaemon/proto/runtime_prepare_test.go @@ -13,7 +13,7 @@ import ( ) func capabilityBegin() RuntimePreparePayload { - return RuntimePreparePayload{Step: "begin", EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), + return RuntimePreparePayload{BudgetMS: 300000, Step: "begin", EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "skill", Skill: &agentskill.Metadata{Type: "inline", Name: "example", Description: "Example"}, SizeBytes: 10, SHA256: strings.Repeat("a", 64)} } @@ -23,6 +23,9 @@ func TestCapabilitiesRequestValidation(t *testing.T) { t.Fatal("valid skill refused") } for name, mutate := range map[string]func(*RuntimePreparePayload){ + "missing budget": func(p *RuntimePreparePayload) { p.BudgetMS = 0 }, + "negative budget": func(p *RuntimePreparePayload) { p.BudgetMS = -1 }, + "excessive budget": func(p *RuntimePreparePayload) { p.BudgetMS = RuntimePrepareMaxBudgetMS + 1 }, "noncanonical identity": func(p *RuntimePreparePayload) { p.EnvironmentID = "AAAAAAAA-AAAA-4AAA-8AAA-AAAAAAAAAAAA" }, "zero identity": func(p *RuntimePreparePayload) { p.SessionID = uuid.Nil.String() }, "begin data": func(p *RuntimePreparePayload) { p.Data = []byte("x") }, @@ -91,7 +94,7 @@ func TestCapabilitiesRequestValidation(t *testing.T) { } func TestCapabilitiesFinalizeRequiresExplicitBoundedSelection(t *testing.T) { - p := RuntimePreparePayload{Step: "begin", EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "finalize", Sources: &agentcapabilities.Input{}} + p := RuntimePreparePayload{BudgetMS: 300000, Step: "begin", EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "finalize", Sources: &agentcapabilities.Input{}} if !ValidRuntimePrepareRequest(p) { t.Fatal("explicit empty finalization refused") } @@ -173,7 +176,7 @@ func TestLocalEnvironmentCarriesSelectionButNeverInstalledRoots(t *testing.T) { } func TestRuntimePreparationInitialActions(t *testing.T) { - base := RuntimePreparePayload{Step: "begin", EnvironmentID: uuid.NewString(), SessionID: uuid.NewString()} + base := RuntimePreparePayload{BudgetMS: 300000, Step: "begin", EnvironmentID: uuid.NewString(), SessionID: uuid.NewString()} for _, initialization := range []RuntimeInitialization{ {Action: "configure", Env: map[string]string{"EXAMPLE": "value"}}, {Action: "npm", Packages: []string{"typescript"}}, @@ -263,7 +266,7 @@ func TestRuntimePreparationExitCodes(t *testing.T) { } func TestRuntimePreparationPortableSources(t *testing.T) { - p := RuntimePreparePayload{Step: "begin", EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "finalize", Sources: &agentcapabilities.Input{Directories: []string{`C:\Users\operator\skills`, `\\host\share\skills`}}} + p := RuntimePreparePayload{BudgetMS: 300000, Step: "begin", EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "finalize", Sources: &agentcapabilities.Input{Directories: []string{`C:\Users\operator\skills`, `\\host\share\skills`}}} if !ValidRuntimePrepareRequest(p) { t.Fatal("portable sources rejected by wire") } @@ -271,3 +274,18 @@ func TestRuntimePreparationPortableSources(t *testing.T) { t.Fatal("Windows sources accepted by Linux resolver") } } + +func TestRuntimePreparationBudgetOnlyOnBegin(t *testing.T) { + for _, p := range []RuntimePreparePayload{{Step: "commit", BudgetMS: 1}, {Step: "chunk", Data: []byte("a"), BudgetMS: 1}} { + if ValidRuntimePrepareRequest(p) { + t.Fatal("budget admitted outside begin") + } + } + for _, budget := range []int64{1, RuntimePrepareMaxBudgetMS} { + p := capabilityBegin() + p.BudgetMS = budget + if !ValidRuntimePrepareRequest(p) { + t.Fatal("valid budget boundary rejected", budget) + } + } +} diff --git a/internal/agentdaemon/proto/suspend.go b/internal/agentdaemon/proto/suspend.go index 053b6fa46..b9c4dede7 100644 --- a/internal/agentdaemon/proto/suspend.go +++ b/internal/agentdaemon/proto/suspend.go @@ -17,6 +17,10 @@ type EnvironmentSuspendPayload struct { SuspendID string `json:"suspend_id"` } +// An accepted quiesce confirms closed admission, drained work and receipts, +// and retention of settled idle native Executors with idle expiry fenced until +// exact authenticated resume. It does not confirm a filesystem checkpoint or +// that provider-owned compute has stopped. type EnvironmentSuspendResultPayload struct { EnvironmentID string `json:"environment_id"` SuspendID string `json:"suspend_id"` diff --git a/internal/agentdaemon/proto/version.go b/internal/agentdaemon/proto/version.go index 524883ee6..2dcbf7337 100644 --- a/internal/agentdaemon/proto/version.go +++ b/internal/agentdaemon/proto/version.go @@ -2,7 +2,7 @@ package proto // Version identifies the complete Core–Runtime wire contract. Change it when // removing or changing a payload or its semantics; deploy both endpoints together. -const Version = "0.13.0" +const Version = "0.16.0" // VersionCompatible accepts only this contract. Patch drift, prerelease suffixes // and malformed versions do not select an implicit compatibility path. diff --git a/scripts/build-claude-runtime.sh b/scripts/build-claude-runtime.sh index af5b3b587..d5b9c60a8 100755 --- a/scripts/build-claude-runtime.sh +++ b/scripts/build-claude-runtime.sh @@ -22,6 +22,7 @@ node "$repo_root/scripts/check-claude-sdk-runtime.mjs" "$context/claude-sdk" -o "$context/oac-daemon" ./apps/daemon/cmd/oac-daemon ) cp "$repo_root/services/core/deploy/claude/Dockerfile" "$context/Dockerfile" +cp "$repo_root/services/core/deploy/runtime_profile.py" "$context/runtime_profile.py" mkdir -p "$output_dir" cp -R "$context/." "$output_dir/" printf 'Claude Runtime image context: %s\n' "$output_dir" diff --git a/scripts/build-codex-runtime.sh b/scripts/build-codex-runtime.sh index fcae07a81..8269cd07c 100755 --- a/scripts/build-codex-runtime.sh +++ b/scripts/build-codex-runtime.sh @@ -33,6 +33,7 @@ trap 'rm -rf "$context"' EXIT cp "$native_dir/bin/codex" "$context/codex" cp -R "$native_dir/codex-resources" "$context/codex-resources" cp "$repo_root/services/core/deploy/codex/Dockerfile" "$context/Dockerfile" +cp "$repo_root/services/core/deploy/runtime_profile.py" "$context/runtime_profile.py" # Preserve the previous bundle if compilation or validation failed. mkdir -p "$output_dir" cp -R "$context/." "$output_dir/" diff --git a/scripts/build-core-distribution.sh b/scripts/build-core-distribution.sh index ac40a609a..37da251f5 100755 --- a/scripts/build-core-distribution.sh +++ b/scripts/build-core-distribution.sh @@ -110,16 +110,12 @@ cp services/core/deploy/codex/seccomp.json "$bundle/runtime/" cp LICENSE "$bundle/" OAC_DEV_BUILD_REVISION="$revision" scripts/build-core-image-context.sh "$stage/core" -( - cd services/core/tools/microsandbox-provider - GOWORK=off CGO_ENABLED=1 go build -mod=readonly -trimpath \ - -o "$stage/core/bin/oac-microsandbox-provider" . -) -msb_archive="${CORE_DISTRIBUTION_MICROSANDBOX_ARCHIVE:-$runtime_root/cache/microsandbox-v0.7.2-linux-x86_64.tar.gz}" +python3 scripts/build-microsandbox-provider.py build "$stage/core/bin/oac-microsandbox-provider" +msb_archive="${CORE_DISTRIBUTION_MICROSANDBOX_ARCHIVE:-$runtime_root/cache/microsandbox-v0.7.8-linux-x86_64.tar.gz}" if [[ ! -f "$msb_archive" ]]; then mkdir -p "$(dirname "$msb_archive")" curl --fail --location --proto '=https' --tlsv1.2 \ - https://github.com/superradcompany/microsandbox/releases/download/v0.7.2/microsandbox-linux-x86_64.tar.gz \ + https://github.com/superradcompany/microsandbox/releases/download/v0.7.8/microsandbox-linux-x86_64.tar.gz \ --output "$stage/microsandbox.download" python3 scripts/core-distribution-manifest.py extract-runtime "$stage/microsandbox.download" "$stage/core/microsandbox" mv "$stage/microsandbox.download" "$msb_archive" diff --git a/scripts/build-mcode-runtime.sh b/scripts/build-mcode-runtime.sh index 04a507698..8698a2212 100644 --- a/scripts/build-mcode-runtime.sh +++ b/scripts/build-mcode-runtime.sh @@ -23,6 +23,7 @@ cp -RL "$companion/." "$context/mcode-harness/" -o "$context/oac-daemon" ./apps/daemon/cmd/oac-daemon ) cp "$repo_root/services/core/deploy/mcode/Dockerfile" "$context/Dockerfile" +cp "$repo_root/services/core/deploy/runtime_profile.py" "$context/runtime_profile.py" mkdir -p "$output" cp -R "$context/." "$output/" printf 'MiniMax Code Runtime image context: %s\n' "$output" diff --git a/scripts/build-microsandbox-provider.py b/scripts/build-microsandbox-provider.py new file mode 100644 index 000000000..2e5e31f90 --- /dev/null +++ b/scripts/build-microsandbox-provider.py @@ -0,0 +1,90 @@ +#!/usr/bin/env python3 +"""Build the provider with the official SDK source and matching embedded FFI.""" + +import argparse +import hashlib +import json +import os +from pathlib import Path +import re +import shutil +import subprocess +import tempfile + +SDK_MODULE = "github.com/superradcompany/microsandbox/sdk/go" +FFI_SHA256 = "bc079888050a92d3652191ae8e18fba96e1ac66dc78bce4e5aaeb7eca9b04e25" +FFI_FILE = "libmicrosandbox_go_ffi-linux-amd64.so" + + +def verify_ffi(path): + with Path(path).open("rb") as stream: + digest = hashlib.sha256() + for block in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(block) + if digest.hexdigest() != FFI_SHA256: + raise ValueError("Official microsandbox FFI checksum mismatch") + + +def install_ffi(source, destination): + verify_ffi(source) + if destination.is_symlink() or not destination.is_file() or destination.stat().st_size != 0: + raise ValueError("Official SDK must contain an empty FFI release sentinel") + shutil.copyfile(source, destination) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("operation", choices=("build", "test")) + parser.add_argument("output", nargs="?", type=Path) + args = parser.parse_args() + if (args.operation == "build") != (args.output is not None): + parser.error("build requires an absolute output path; test takes no output") + if args.output is not None and not args.output.is_absolute(): + parser.error("output must be absolute") + root = Path(__file__).resolve().parents[1] + module = root / "services/core/tools/microsandbox-provider" + env = dict(os.environ, GOWORK="off", CGO_ENABLED="1") + def run(*command, cwd=module): + try: + return subprocess.check_output(command, cwd=cwd, env=env, text=True) + except subprocess.CalledProcessError as error: + print(error.output, end="") + raise + if run("go", "env", "GOOS", "GOARCH").splitlines() != ["linux", "amd64"]: + parser.error("the provider distribution requires Linux amd64") + run("go", "mod", "download", SDK_MODULE) + sdk = json.loads(run("go", "list", "-mod=readonly", "-m", "-json", SDK_MODULE)) + if "Replace" in sdk: + raise ValueError("The SDK must be the pinned official module") + versions = re.findall(r'^const sdkVersion = "([0-9.]+)"$', (Path(sdk["Dir"]) / "setup.go").read_text(), re.M) + if len(versions) != 1: + raise ValueError("Official SDK release declaration changed") + version = versions[0] + cache = Path(os.environ.get("OAC_DEV_HOME", str(Path.home() / ".oac"))) / "cache" + if not cache.is_absolute(): + raise ValueError("OAC_DEV_HOME must be absolute") + cache.mkdir(parents=True, exist_ok=True) + ffi = cache / (FFI_SHA256 + ".so") + with tempfile.TemporaryDirectory(prefix="oac-microsandbox-build-", dir=cache) as temporary: + stage = Path(temporary) + if not ffi.exists(): + download = stage / FFI_FILE + run("curl", "--fail", "--location", "--silent", "--show-error", "--retry", "3", "--output", str(download), + f"https://github.com/superradcompany/microsandbox/releases/download/v{version}/{FFI_FILE}") + verify_ffi(download) + os.replace(download, ffi) + verify_ffi(ffi) + for source in [*module.glob("*.go"), module / "go.mod", module / "go.sum"]: + shutil.copyfile(source, stage / source.name) + run("go", "mod", "vendor", "-o", str(stage / "vendor")) + install_ffi(ffi, stage / "vendor" / SDK_MODULE / "internal/bundle/bundles" / FFI_FILE) + if args.operation == "test": + print(run("go", "test", "-mod=vendor", "./...", "-count=1", cwd=stage), end="") + print(run("go", "vet", "-mod=vendor", "./...", cwd=stage), end="") + else: + args.output.parent.mkdir(parents=True, exist_ok=True) + print(run("go", "build", "-mod=vendor", "-trimpath", "-o", str(args.output), ".", cwd=stage), end="") + + +if __name__ == "__main__": + main() diff --git a/scripts/build-microsandbox-provider.test.py b/scripts/build-microsandbox-provider.test.py new file mode 100644 index 000000000..d53f80ccd --- /dev/null +++ b/scripts/build-microsandbox-provider.test.py @@ -0,0 +1,48 @@ +#!/usr/bin/env python3 +import hashlib +import importlib.util +from pathlib import Path +import tempfile +import unittest +from unittest.mock import patch + +spec = importlib.util.spec_from_file_location("build_provider", Path(__file__).with_name("build-microsandbox-provider.py")) +build = importlib.util.module_from_spec(spec) +spec.loader.exec_module(build) + + +class OfficialBundleTests(unittest.TestCase): + def test_only_verified_payload_replaces_empty_sentinel(self): + with tempfile.TemporaryDirectory() as directory: + source, destination = Path(directory) / "ffi", Path(directory) / "sentinel" + source.write_bytes(b"official ffi") + destination.touch() + with patch.object(build, "FFI_SHA256", hashlib.sha256(source.read_bytes()).hexdigest()): + build.install_ffi(source, destination) + self.assertEqual(destination.read_bytes(), source.read_bytes()) + with self.assertRaisesRegex(ValueError, "empty FFI"): + build.install_ffi(source, destination) + + def test_corrupt_payload_does_not_replace_sentinel(self): + with tempfile.TemporaryDirectory() as directory: + source, destination = Path(directory) / "ffi", Path(directory) / "sentinel" + source.write_bytes(b"wrong release") + destination.touch() + with self.assertRaisesRegex(ValueError, "checksum mismatch"): + build.install_ffi(source, destination) + self.assertEqual(destination.read_bytes(), b"") + + def test_symlink_sentinel_is_rejected(self): + with tempfile.TemporaryDirectory() as directory: + source, destination, outside = (Path(directory) / name for name in ("ffi", "sentinel", "outside")) + source.write_bytes(b"official ffi") + outside.touch() + destination.symlink_to(outside) + with patch.object(build, "FFI_SHA256", hashlib.sha256(source.read_bytes()).hexdigest()): + with self.assertRaisesRegex(ValueError, "empty FFI"): + build.install_ffi(source, destination) + self.assertEqual(outside.read_bytes(), b"") + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/ci_plan.py b/scripts/ci_plan.py index 02f0f47f6..19a0fadd2 100644 --- a/scripts/ci_plan.py +++ b/scripts/ci_plan.py @@ -78,6 +78,7 @@ (("scripts/build-native-", "scripts/native-"), SCRIPTS, ("native", "backend", "distribution")), (("scripts/build-core.sh", "scripts/build-core-image-context.sh"), (".sh",), ("backend", "api", "distribution", "native")), (("deploy/distribution/",), ("Dockerfile",), ("backend", "api", "distribution", "native", "compose")), + (("scripts/build-microsandbox-provider.",), (".py",), ("backend", "distribution")), (("scripts/build-e2b-provider.sh",), (".sh",), ("backend", "api", "distribution")), (("scripts/build-claude", "scripts/check-claude", "scripts/build-mcode", "scripts/prepare-release-runtimes.sh"), SCRIPTS, ("harness", "native", "backend", "distribution")), diff --git a/scripts/ci_plan_test.py b/scripts/ci_plan_test.py index d5ee84204..40ec23043 100644 --- a/scripts/ci_plan_test.py +++ b/scripts/ci_plan_test.py @@ -57,6 +57,10 @@ def test_generated_outputs_keep_freshness_checks(self): with self.subTest(path=path): self.assertIn("distribution", self.jobs(path)) + def test_microsandbox_builder_keeps_helper_and_distribution_checks(self): + for path in ("scripts/build-microsandbox-provider.py", "scripts/build-microsandbox-provider.test.py"): + self.assertTrue({"backend", "distribution"} <= self.jobs(path)) + def test_distribution_image_build_keeps_api_acceptance(self): plan = ci.select(["scripts/build-core-distribution.sh"]) self.assertTrue(plan["image"]) diff --git a/scripts/core-distribution-manifest.py b/scripts/core-distribution-manifest.py index 3529fb8fe..cdb7e9843 100644 --- a/scripts/core-distribution-manifest.py +++ b/scripts/core-distribution-manifest.py @@ -17,7 +17,7 @@ import zipapp -RUNTIME_ARCHIVE_SHA256 = "47c223e3ef5298abf05f47ed9f87981106e400d99bb3f1d042d4d6881346b18b" +RUNTIME_ARCHIVE_SHA256 = "86f9f72dc3e639c7175bc07909b4b63ce412517c1ef8a2e1921171af5682fded" DIGEST = re.compile(r"sha256:[0-9a-f]{64}\Z") sys.path.insert(0, str(pathlib.Path(__file__).resolve().parents[1] / "deploy/node")) import provider_assets @@ -204,7 +204,7 @@ def verify_runtime(image, daemon, source): def extract_runtime(archive, destination): if sha256(archive) != RUNTIME_ARCHIVE_SHA256: - raise ValueError("microsandbox v0.7.2 release checksum mismatch") + raise ValueError("microsandbox v0.7.8 release checksum mismatch") destination = pathlib.Path(destination) destination.mkdir(parents=True, exist_ok=True) with tarfile.open(archive, "r:gz") as bundle: @@ -346,7 +346,7 @@ def node_payload(bundle, stage, revision, source_tree, artifact_base_url="", off "image_manifest_digests": {name: identity[1] for name, identity in identities.items()}, "runtime_ref": "oac-runtime@" + digest, "microsandbox": { - "version": "0.7.2", + "version": "0.7.8", "runtime_sha256": sha256(stage / "core/microsandbox/msb"), "firmware_sha256": sha256(stage / "core/microsandbox/libkrunfw.so.5.6.1"), }, diff --git a/scripts/name-allowlist.json b/scripts/name-allowlist.json index c29a4637e..69b3837f1 100644 --- a/scripts/name-allowlist.json +++ b/scripts/name-allowlist.json @@ -263,5 +263,10 @@ "path": "scripts/ci_plan_test.py", "regex": "(?:example/parsar|@oac/parsar-example)", "reason": "The named Parsar example owns its dependency lock and component checks." + }, + { + "path": "apps/daemon/internal/cli/connect_suspend_test.go", + "regex": "\"agents-api-\"", + "reason": "The cleanup regression uses the persisted Session state-key prefix required by Runtime executor ownership validation." } ] diff --git a/services/core/IMPLEMENTATION.md b/services/core/IMPLEMENTATION.md index 7f3864996..f60275a1c 100644 --- a/services/core/IMPLEMENTATION.md +++ b/services/core/IMPLEMENTATION.md @@ -35,10 +35,10 @@ Domain owners, each with its PostgreSQL adapter under `internal/persistence/post - `environmenttemplates` (`templatepg`): Environment Templates, their validation and default network, their sealed setup, initial files, Skills and Plugins, and the resolved Template that Session creation composes into its Environment. - `modelconfiguration` (`modelconfigurationpg`): each Harness's deployment default model configuration and its last-use observations. - `skills` (`skillpg`): Skills and their immutable versions: archive checks, the default and latest pointers, version selection and deletion, and each version's sealed archive. Session creation freezes selected versions inside its own transaction: `skillpg.LockSkills` locks the selected Skills in ID order, `skillpg.ReadVersionForFreeze` opens and verifies a version, and `sessions` selects each version with the `skills` rules. -- `deployment` and its subpackage `deployment/placement` (`deploymentpg`, `placementpg`): the sandbox deployment and its nodes: provider configuration and the sealed credential, the specification and retained generations, setup, update and switch, node enrollment, identity and authentication, generation configuration, capacity, presence, status, host history, reset, and the counts of nodes and sandboxes bound to the public address; and hosted runtime allocations: reservation with the dedicated device, compute ownership and settlement, observation diagnostics, activity, compute phases, wake receipts and cleanup, with the reads that schedule, discover and authorize them. It interprets Sandbox Provider declarations through the `providers.Registry` it is given, whose lookups return typed errors. `cmd/server` builds that registry and calls it only to build a direct Provider and to discover a Provider's configuration; it takes each setup's mode and declared operations from the `Setup` that `deployment` returns. `deployment/placement` owns hosted admission and placement: `cmd/server` builds one `placement.Rules` from the registry and the public URL, both fixed while Core runs, and gives it to `deployment.Service` and `sessions.Service`; its pure decisions admit a hosted Session, choose its node, and admit an allocation's reserved node and a restore on it, and its errors keep one status, code and message through `writeSandboxError`. `placementpg` loads the facts those decisions read and applies them on the caller's transaction-bound queries; it has no Store or transaction runner. The only other caller is Session creation: `sessions` decides admission and placement with those rules over the `placementpg` participants inside the creation transaction. Deployment changes and allocation writes run through `deployment.ExecutionOperations` on `deploymentpg.NewExecution(lease, …)`, which the Worker receives as `execution.Owner.Deployment`. Each allocation write locks the owning Session, decides in `deployment`, applies in `deploymentpg` and prunes the Session's change journal in the same leased transaction; cleanup settles the Session through the `sessions` procedures on a `sessionpg.SessionTx` bound to that transaction. The Worker reads the deployment, prepares a selection's setup and records live-compute activity through the pooled `deployment.Service` in `execution.Dispatcher.Deployment`, and lists the Sessions a reset still has to archive, schedules node lifecycles and reads allocations through the pooled `deployment.Reader` in `execution.Dispatcher.DeploymentReader`; a pooled read carries no lease, and the leased write that follows it rechecks the allocation's owner. Node management and reads use the pooled `deploymentpg.Store`, which `cmd/server` also reads the owner epoch from and runs the host-history sampler on. Provider calls run outside transactions, and the final transaction rechecks the expected generation. Session archive crosses the Session and the deployment, so it is a `deployment.ExecutionOperations` operation: `ArchiveSession`, which the administrator's archive route calls through `api.Execution.SessionArchive`, and `ArchiveResetSession`, which a reset calls, lock the Session through `deploymentpg`'s `WithSessionArchive`, check the deployment's generation and the running reset in `deployment`, then expire the Environment and cancel its work through the `sessions` procedures, revoke the allocation's device and request its cleanup, and record the audit entry in one leased transaction. `deployment.ObservationResolver` resolves a Session's Runtime observation target for `runtimeobs` from `sessions.SessionReader` and `deployment.Reader`. +- `deployment` and its subpackage `deployment/placement` (`deploymentpg`, `placementpg`): the sandbox deployment and its nodes: provider configuration and the sealed credential, the specification and retained generations, setup, update and switch, node enrollment, identity and authentication, generation configuration, capacity, presence, status, host history, reset, and the counts of nodes and sandboxes bound to the public address; and hosted runtime allocations: reservation with the dedicated device, compute ownership and settlement, observation diagnostics, activity, compute phases, wake receipts and cleanup, with the reads that schedule, discover and authorize them. It interprets Sandbox Provider declarations through the `providers.Registry` it is given, whose lookups return typed errors. `cmd/server` builds that registry and calls it only to build a direct Provider and to discover a Provider's configuration; it takes each setup's mode and declared operations from the `Setup` that `deployment` returns. `deployment/placement` owns hosted admission and placement: `cmd/server` builds one `placement.Rules` from the registry and the public URL, both fixed while Core runs, and gives it to `deployment.Service` and `sessions.Service`; its pure decisions admit a hosted Session, choose its node, and admit an allocation's reserved node and a restore on it, and its errors keep one status, code and message through `writeSandboxError`. `placementpg` loads the facts those decisions read and applies them on the caller's transaction-bound queries; it has no Store or transaction runner. Session creation uses those rules and the `placementpg` participants only for durable admission inside the creation transaction; it does not reserve compute. The leased execution manager owns the common placement entry point for first compute and replacement compute through `deployment.ExecutionOperations.EnsurePlacement`. Deployment changes and allocation writes run through `deployment.ExecutionOperations` on `deploymentpg.NewExecution(lease, …)`, which the Worker receives as `execution.Owner.Deployment`. Each allocation write locks the owning Session, decides in `deployment`, applies in `deploymentpg` and prunes the Session's change journal in the same leased transaction; cleanup settles the Session through the `sessions` procedures on a `sessionpg.SessionTx` bound to that transaction. The Worker reads the deployment, prepares a selection's setup and records live-compute activity through the pooled `deployment.Service` in `execution.Dispatcher.Deployment`, and lists the Sessions a reset still has to archive, schedules node lifecycles and reads allocations through the pooled `deployment.Reader` in `execution.Dispatcher.DeploymentReader`; a pooled read carries no lease, and the leased write that follows it rechecks the allocation's owner. Node management and reads use the pooled `deploymentpg.Store`, which `cmd/server` also reads the owner epoch from and runs the host-history sampler on. Provider calls run outside transactions, and the final transaction rechecks the expected generation. Session archive crosses the Session and the deployment, so it is a `deployment.ExecutionOperations` operation: `ArchiveSession`, which the administrator's archive route calls through `api.Execution.SessionArchive`, and `ArchiveResetSession`, which a reset calls, lock the Session through `deploymentpg`'s `WithSessionArchive`, check the deployment's generation and the running reset in `deployment`, then expire the Environment and cancel its work through the `sessions` procedures, revoke the allocation's device and request its cleanup, and record the audit entry in one leased transaction. `deployment.ObservationResolver` resolves a Session's Runtime observation target for `runtimeobs` from `sessions.SessionReader` and `deployment.Reader`. - `coremetrics` (`coremetricspg`): the Core metrics PostgreSQL holds: the root Turn queue counts, the root Turn history, read from one read-only snapshot, and the database size. `cmd/server`'s Core metrics source adds them to the process, pool, Worker and daemon registry measurements. - `runtimehistory` (`runtimehistorypg`): Runtime history samples: the periodic export, scoped reads and retention, which also prunes node-host samples. `runtimehistory.Service` scopes a read to the Session's hosted Environment through `sessions.EnvironmentReader`. -- `sessions` (`sessionpg`): Session use cases and reads. The pooled `sessions.Service` runs the use cases on `sessionpg.Store`, which implements `sessions.Storage`, and plain reads use `sessions.Reader`, which `sessionpg.Store` also implements, directly. These cover Sessions (their creation, reads, the change journal and stream snapshot, diagnostics, the frozen execution configuration, measured usage and archive state, metadata updates, deletion and the public write audit), Turns (their reads and the execution work scan), input admission (a public input batch, Environment input reservation and the expiry of one reservation, with the reads of a Turn's admitted inputs, a reservation and the Environment input work scan), the model provider a Session froze, which `sessionpg.Store` opens with the credential key, root Items and Subagents (their reads), Session Artifacts (their reads, deletion and the staging of a Turn's export), Environments (their reads, the initialization list and the frozen setup and initial files, which `sessionpg.Store` opens with the credential key), devices (their reads, which include a Session's execution binding and the execution device list, their creation, revocation, heartbeats and Runtime enrollment, and the archived cancellation receipt the daemon gateway reads), the administrator's views across Projects (a Project's asset counts and Sessions for the summary, and the Sessions whose Runtime the administrator observes), executor credentials (authentication and a Project's credential state through `sessions.ExecutorCredentialReader`, and issuance, rotation and revocation; the Core-key Project operations record their audit entry in the same transaction) and native installation authorization and claims, whose tokens `sessionpg.Store` signs with the credential key. Session creation runs in one pooled `sessions.CreationTx`: `sessions` validates the request, computes its retry identity, with the provider key fingerprinted by `sessionpg.Store` under the credential key, and orders the upsert, hosted admission, the Skill freeze, the sealed model provider, execution configuration, initial files and setup, the Environment, placement and the initial input; `FindSessionCreation` finds an earlier creation by its recorded intent. `cmd/server` builds one `sessionpg.Store` with the credential key and the Service with its `placement.Rules`, and wires the Service and the Store into `execution.Dispatcher.Sessions` and `SessionsReader`, into the api fields, into Runtime enrollment and into the daemon gateway; the Worker creates Sessions through the Service after its execution checks, waking the scheduler only after the commit, and stages Artifacts through it; `cmd/environment-key` builds one without the key, and the Service without placement rules, for its credential commands. Turn transitions, execution completion, which alone publishes or discards a Turn's staged Artifacts, the start of Artifact capture, function calls and their application receipts, the Turn execution journal, Environment initialization, connection observations and their reconciliation, Session device binding, file-write reservation and settlement, and the promotion, failure and bulk expiry of Environment input reservations run through `sessions.ExecutionOperations` on `sessionpg.NewExecution(lease)`, which the Worker receives as `execution.Owner.Sessions`. +- `sessions` (`sessionpg`): Session use cases and reads. The pooled `sessions.Service` runs the use cases on `sessionpg.Store`, which implements `sessions.Storage`, and plain reads use `sessions.Reader`, which `sessionpg.Store` also implements, directly. These cover Sessions (their creation, reads, the change journal and stream snapshot, diagnostics, the frozen execution configuration, measured usage and archive state, metadata updates, deletion and the public write audit), Turns (their reads and the execution work scan), input admission (a public input batch, Environment input reservation and the expiry of one reservation, with the reads of a Turn's admitted inputs, a reservation and the Environment input work scan), the model provider a Session froze, which `sessionpg.Store` opens with the credential key, root Items and Subagents (their reads), Session Artifacts (their reads, deletion and the staging of a Turn's export), Environments (their reads, the initialization list and the frozen setup and initial files, which `sessionpg.Store` opens with the credential key), devices (their reads, which include a Session's execution binding and the execution device list, their creation, revocation, heartbeats and Runtime enrollment, and the archived cancellation receipt the daemon gateway reads), the administrator's views across Projects (a Project's asset counts and Sessions for the summary, and the Sessions whose Runtime the administrator observes), executor credentials (authentication and a Project's credential state through `sessions.ExecutorCredentialReader`, and issuance, rotation and revocation; the Core-key Project operations record their audit entry in the same transaction) and native installation authorization and claims, whose tokens `sessionpg.Store` signs with the credential key. Session creation runs in one pooled `sessions.CreationTx`: `sessions` validates the request, computes its retry identity, with the provider key fingerprinted by `sessionpg.Store` under the credential key, and orders the upsert, hosted admission, the Skill freeze, the sealed model provider, execution configuration, initial files and setup, the Environment and the initial input; `FindSessionCreation` finds an earlier creation by its recorded intent. `cmd/server` builds one `sessionpg.Store` with the credential key and the Service with its `placement.Rules`, and wires the Service and the Store into `execution.Dispatcher.Sessions` and `SessionsReader`, into the api fields, into Runtime enrollment and into the daemon gateway; the Worker creates Sessions through the Service after its execution checks, waking the scheduler only after the commit, and stages Artifacts through it; `cmd/environment-key` builds one without the key, and the Service without placement rules, for its credential commands. Turn transitions, execution completion, which alone publishes or discards a Turn's staged Artifacts, the start of Artifact capture, function calls and their application receipts, the Turn execution journal, Environment initialization, connection observations and their reconciliation, Session device binding, file-write reservation and settlement, and the promotion, failure and bulk expiry of Environment input reservations run through `sessions.ExecutionOperations` on `sessionpg.NewExecution(lease)`, which the Worker receives as `execution.Owner.Sessions`. ## Request handling diff --git a/services/core/cmd/sandbox-node/main.go b/services/core/cmd/sandbox-node/main.go index 2a6d3b032..4e955be99 100644 --- a/services/core/cmd/sandbox-node/main.go +++ b/services/core/cmd/sandbox-node/main.go @@ -93,8 +93,12 @@ func run(ctx context.Context, args []string) error { } expected := node.Identity{SpecificationDigest: built.SpecificationDigest, DeploymentGeneration: config.Generation, InstallationID: built.InstallationID, Provider: config.Provider, BackendFingerprint: built.BackendFingerprint} probe := func(ctx context.Context) (node.Health, error) { - err := built.Probe(ctx) - return node.Health{ProviderReady: err == nil}, err + checkpoint, err := built.Probe(ctx) + status := sandbox.GenerationStatus{Generation: config.Generation, SpecificationDigest: built.SpecificationDigest, State: "ready", Checkpoint: checkpoint} + if err != nil { + status.State, status.Diagnostic, status.Checkpoint = "failed", sandbox.NodeDiagnostic(err), nil + } + return node.Health{ProviderReady: err == nil, Generations: []sandbox.GenerationStatus{status}}, err } if args[0] == "register" { if !filepath.IsAbs(*tokenFile) { diff --git a/services/core/cmd/server/main.go b/services/core/cmd/server/main.go index 27b972ae2..2969cdaa9 100644 --- a/services/core/cmd/server/main.go +++ b/services/core/cmd/server/main.go @@ -235,7 +235,7 @@ func run(config processconfig.Config) error { if err != nil { return err } - deploymentExecution, err := deployment.NewExecutionOperations(deploymentService, deploymentpg.NewExecution(lease, credentialKey)) + deploymentExecution, err := deployment.NewExecutionOperations(deploymentService, deploymentpg.NewExecution(lease, credentialKey), dispatcher.Engines) if err != nil { return errors.Join(err, lease.Close(ctx)) } diff --git a/services/core/cmd/server/managed_nodes.go b/services/core/cmd/server/managed_nodes.go index 78a5639f7..402a1912c 100644 --- a/services/core/cmd/server/managed_nodes.go +++ b/services/core/cmd/server/managed_nodes.go @@ -67,7 +67,7 @@ func configureManagedNodes(nodes *deployment.Service, reader deployment.Reader, if err := owner(ctx); err != nil { return err } - return nodes.Heartbeat(ctx, n.NodeID, connection, epoch, nodeHealthRecord(health)) + return nodes.Heartbeat(ctx, n.NodeID, connection, epoch, nodeHealthRecord(health), health.Generations) }, }) result.setup = &managedSetup{capacity: config.SandboxCapacity, processPaths: config.ProviderPaths, registry: registry, deployment: nodes, allocations: reader, hub: result.hub, installationID: config.InstallationID, runtimeAPI: config.PublicOrigin.RuntimeAPI()} diff --git a/services/core/deploy/claude/Dockerfile b/services/core/deploy/claude/Dockerfile index 115878e19..9862c8582 100644 --- a/services/core/deploy/claude/Dockerfile +++ b/services/core/deploy/claude/Dockerfile @@ -5,6 +5,8 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ ca-certificates bash git python3 python3-pip ripgrep \ && rm -rf /var/lib/apt/lists/* \ && mkdir -p /environment/workspace /workspace /home/runtime +COPY runtime_profile.py /tmp/oac-runtime-profile.py +RUN python3 /tmp/oac-runtime-profile.py && rm /tmp/oac-runtime-profile.py COPY --chmod=0555 oac-daemon /usr/local/bin/ COPY claude-sdk /opt/claude-sdk ENV HOME=/home/runtime OAC_RUNTIME_HOME=/home/runtime/.oac \ @@ -12,7 +14,8 @@ ENV HOME=/home/runtime OAC_RUNTIME_HOME=/home/runtime/.oac \ OAC_RUNTIME_CLAUDE_SDK_ENTRYPOINT=/opt/claude-sdk/dist/main.js \ OAC_RUNTIME_WORKSPACE=/environment/workspace \ OAC_RUNTIME_INITIALIZATION_DIRECTORY=/environment/initialization \ - OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages + OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages \ + OAC_RUNTIME_STATE_DIRECTORY=/environment/runtime-state USER 1000:1000 RUN node /opt/claude-sdk/dist/runtime_check.js /opt/claude-sdk/dist/main.js WORKDIR /environment/workspace diff --git a/services/core/deploy/claude/README.md b/services/core/deploy/claude/README.md index 327af0f16..f790d50b2 100644 --- a/services/core/deploy/claude/README.md +++ b/services/core/deploy/claude/README.md @@ -14,7 +14,7 @@ Claude accepts only the `anthropic` model protocol ([`harnessconfig/claudesdk`]( ## Native state and recovery -With a workspace binding, the SDK's history, home and scratch directories are `history`, `home` and `scratch` under `$OAC_RUNTIME_HOME/runtime/claude-sdk/` (mode 0700; `OAC_RUNTIME_HOME` defaults to `~/.oac`). The daemon's authentication stays under `$OAC_RUNTIME_HOME/daemon/`. These locations keep Session state apart; they do not restrict the tools. +With a workspace binding, the SDK stores retained history at `/runtime/claude-sdk/history`, where the Runtime supplies `` from the [state directory setting](../../../../docs/configuration.md#runtime-resource-directories). Without a workspace binding, native state uses `/daemon//runtime/claude-sdk`. Private launch home and scratch directories for workspace execution stay at `$OAC_RUNTIME_HOME/runtime/claude-sdk/home` and `$OAC_RUNTIME_HOME/runtime/claude-sdk/scratch` (mode 0700). The daemon's authentication stays under `$OAC_RUNTIME_HOME/daemon/`. These locations keep native history, temporary process state and device credentials separate; they do not restrict the tools. Recovery uses the SDK's history APIs ([`recovery.ts`](../../../../packages/claude-sdk-adapter/src/recovery.ts)). An explicitly supplied native Session ID must exist. When Core requires existing history but has no recorded ID, the adapter accepts only a single native Session whose recorded cwd equals the bound workspace and which has at least one message. Missing, foreign, ambiguous or empty history is rejected before any model input. @@ -43,7 +43,7 @@ The bridge ([`native_failure.ts`](../../../../packages/claude-sdk-adapter/src/na | Base | Digest-pinned `node:22.23.1-bookworm-slim` with `ca-certificates`, `bash`, `git`, `python3`, `python3-pip` and `ripgrep` | | Programs | `/usr/local/bin/oac-daemon` and the SDK bundle at `/opt/claude-sdk` | | User | UID/GID 1000 with `HOME=/home/runtime` | -| Environment | `OAC_RUNTIME_HOME=/home/runtime/.oac`, `OAC_RUNTIME_CLAUDE_SDK_NODE=/usr/local/bin/node`, `OAC_RUNTIME_CLAUDE_SDK_ENTRYPOINT=/opt/claude-sdk/dist/main.js`, `OAC_RUNTIME_WORKSPACE=/environment/workspace`, `OAC_RUNTIME_INITIALIZATION_DIRECTORY=/environment/initialization`, `OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages` | +| Environment | `OAC_RUNTIME_HOME=/home/runtime/.oac`, `OAC_RUNTIME_CLAUDE_SDK_NODE=/usr/local/bin/node`, `OAC_RUNTIME_CLAUDE_SDK_ENTRYPOINT=/opt/claude-sdk/dist/main.js`, `OAC_RUNTIME_WORKSPACE=/environment/workspace`, `OAC_RUNTIME_INITIALIZATION_DIRECTORY=/environment/initialization`, `OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages`, `OAC_RUNTIME_STATE_DIRECTORY=/environment/runtime-state` | | Entry point | `oac-daemon connect --profile default`, working directory `/environment/workspace` | The build runs the bundle's `runtime_check.js` against its entry point. The distribution copies `/opt/claude-sdk` into the combined Runtime image. Sandboxes run the image with the [Docker sandbox settings](../../../../docs/sandbox-provider.md#docker-adapter). diff --git a/services/core/deploy/codex/Dockerfile b/services/core/deploy/codex/Dockerfile index e79a93cd5..53887748d 100644 --- a/services/core/deploy/codex/Dockerfile +++ b/services/core/deploy/codex/Dockerfile @@ -8,13 +8,16 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ ca-certificates bash git python3 python3-pip ripgrep \ && rm -rf /var/lib/apt/lists/* \ && mkdir -p /environment/workspace /workspace /home/runtime +COPY runtime_profile.py /tmp/oac-runtime-profile.py +RUN python3 /tmp/oac-runtime-profile.py && rm /tmp/oac-runtime-profile.py COPY --chmod=0555 oac-daemon codex /usr/local/bin/ COPY codex-resources /usr/local/codex-resources ENV HOME=/home/runtime OAC_RUNTIME_HOME=/home/runtime/.oac \ OAC_RUNTIME_CODEX_BIN=/usr/local/bin/codex \ OAC_RUNTIME_WORKSPACE=/environment/workspace \ OAC_RUNTIME_INITIALIZATION_DIRECTORY=/environment/initialization \ - OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages + OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages \ + OAC_RUNTIME_STATE_DIRECTORY=/environment/runtime-state USER 1000:1000 RUN test "$(codex --version)" = "codex-cli 0.153.4" WORKDIR /environment/workspace diff --git a/services/core/deploy/codex/README.md b/services/core/deploy/codex/README.md index ac0c9c78b..645908565 100644 --- a/services/core/deploy/codex/README.md +++ b/services/core/deploy/codex/README.md @@ -12,7 +12,7 @@ The adapter accepts only Codex `0.153.4`: installation and recovery checks requi Codex accepts only the `responses` model protocol ([`harnessconfig/codex`](../../../../internal/harnessconfig/codex/configuration.go)). A Chat Completions or Anthropic provider is rejected; nothing converts between protocols. [Model execution](../../../../contracts/agents-api/model-execution.md#deployment-defaults) owns provider selection, including the per-Harness deployment default. -Each Session has its own `CODEX_HOME` at `$OAC_RUNTIME_HOME/daemon/agent-sessions//` (`OAC_RUNTIME_HOME` defaults to `~/.oac`). Native history stays there. The adapter regenerates that directory's `config.toml` on every prompt: the Session's frozen provider bundle becomes the `[model_providers.oac]` block, with `wire_api` `responses`, and the thread is pinned to that provider, so Codex never falls back to its built-in `openai` provider. +Each Session has its own `CODEX_HOME` at `/daemon/agent-sessions//`, where the Runtime supplies `` from the [state directory setting](../../../../docs/configuration.md#runtime-resource-directories). Native history stays there; Runtime device credentials stay in the Runtime home. The state layout alone does not qualify cross-node native continuation. The adapter regenerates that directory's `config.toml` on every prompt: the Session's frozen provider bundle becomes the `[model_providers.oac]` block, with `wire_api` `responses`, and the thread is pinned to that provider, so Codex never falls back to its built-in `openai` provider. Provider bearer credentials use the native `env_key` reference and enter the model-loop child process environment; they are not written into retained `config.toml`. The adapter excludes that variable from native shell environments with `shell_environment_policy.exclude` and disables `shell_snapshot`, whose capture in the pinned native version can otherwise persist the model-loop environment. This retains native thread history without retaining native shell snapshots; it introduces no user setting. ## Execution controls @@ -66,7 +66,7 @@ Every other variant stays unclassified. | Base | Digest-pinned `node:22.23.1-bookworm-slim` with `ca-certificates`, `bash`, `git`, `python3`, `python3-pip` and `ripgrep`, the same base and package layer as the Claude and MiniMax images | | Programs | `/usr/local/bin/oac-daemon`, `/usr/local/bin/codex` (mode 0555) and `/usr/local/codex-resources` | | User | UID/GID 1000 with `HOME=/home/runtime` | -| Environment | `OAC_RUNTIME_HOME=/home/runtime/.oac`, `OAC_RUNTIME_CODEX_BIN=/usr/local/bin/codex`, `OAC_RUNTIME_WORKSPACE=/environment/workspace`, `OAC_RUNTIME_INITIALIZATION_DIRECTORY=/environment/initialization`, `OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages` | +| Environment | `OAC_RUNTIME_HOME=/home/runtime/.oac`, `OAC_RUNTIME_CODEX_BIN=/usr/local/bin/codex`, `OAC_RUNTIME_WORKSPACE=/environment/workspace`, `OAC_RUNTIME_INITIALIZATION_DIRECTORY=/environment/initialization`, `OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages`, `OAC_RUNTIME_STATE_DIRECTORY=/environment/runtime-state` | | Entry point | `oac-daemon connect --profile default`, working directory `/environment/workspace` | The build fails unless `codex --version` reports the pinned version. The image holds no credentials, workspace data or product software. The distribution copies the Codex executable and resources into the combined Runtime image; see the [maintainer guide](../../../../docs/maintainers.md#runtime-images-and-helpers). diff --git a/services/core/deploy/e2b/helper_contract_generated.py b/services/core/deploy/e2b/helper_contract_generated.py index 7f420e0aa..d454cdd8c 100644 --- a/services/core/deploy/e2b/helper_contract_generated.py +++ b/services/core/deploy/e2b/helper_contract_generated.py @@ -17,7 +17,7 @@ REQUEST_FIELDS = ["Version","Operation","Config","Reference","References","Bootstrap","RuntimeBootstrap","Command","Compute","Suspend","Resume","Retained","Deadline"] RESPONSE_FIELDS = ["Version","State","Info","Command","ErrorCode","DeploymentValid","TemplateBuild","Templates","Builds","Observation"] RESUME_FIELDS = ["Workspace","Reference","OperationID","Retained","Target","ReconcileOnly"] -RETAINED_FIELDS = ["Reference","ID","Data","OperationID","SourceGeneration","SourceName","SourceID"] +RETAINED_FIELDS = ["Compatibility","Reference","ID","Data","OperationID","SourceGeneration","SourceName","SourceID"] SDK_VERSION = "2.51.0" SUSPEND_CONTROL_FILE = "/run/oac/daemon-suspend.json" SUSPEND_FIELDS = ["Reference","OperationID","Source","Retained","ReconcileOnly"] diff --git a/services/core/deploy/e2b/init.py b/services/core/deploy/e2b/init.py index f09a1b639..2b2b49dfd 100644 --- a/services/core/deploy/e2b/init.py +++ b/services/core/deploy/e2b/init.py @@ -129,7 +129,7 @@ def initialize(): write_private(ROOT / 'launch.json', identity) environment = prepare_runtime() # Application-owned startup does not participate in Core-managed suspension. - environment.pop("OAC_RUNTIME_DAEMON_SUSPEND_PID_FILE", None) + environment.pop('OAC_RUNTIME_DAEMON_SUSPEND_PID_FILE', None) credential = PROFILE.parent / 'executor-key.json' write_private(credential, payload['executor_key'], owner=1000) source.unlink() diff --git a/services/core/deploy/mcode/Dockerfile b/services/core/deploy/mcode/Dockerfile index 774c1be79..aa43dbbb4 100644 --- a/services/core/deploy/mcode/Dockerfile +++ b/services/core/deploy/mcode/Dockerfile @@ -5,6 +5,8 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ ca-certificates bash git python3 python3-pip ripgrep \ && rm -rf /var/lib/apt/lists/* \ && mkdir -p /environment/workspace /workspace /home/runtime +COPY runtime_profile.py /tmp/oac-runtime-profile.py +RUN python3 /tmp/oac-runtime-profile.py && rm /tmp/oac-runtime-profile.py COPY --chmod=0555 oac-daemon /usr/local/bin/ COPY mcode-harness /opt/mcode-harness ENV HOME=/home/runtime OAC_RUNTIME_HOME=/home/runtime/.oac \ @@ -13,7 +15,8 @@ ENV HOME=/home/runtime OAC_RUNTIME_HOME=/home/runtime/.oac \ OAC_RUNTIME_MCODE_WORKSPACE_BRIDGE=/opt/mcode-harness/bridge.mjs \ OAC_RUNTIME_WORKSPACE=/environment/workspace \ OAC_RUNTIME_INITIALIZATION_DIRECTORY=/environment/initialization \ - OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages + OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages \ + OAC_RUNTIME_STATE_DIRECTORY=/environment/runtime-state USER 1000:1000 RUN node /opt/mcode-harness/check.mjs && /opt/mcode-harness/native/cli.js --version WORKDIR /environment/workspace diff --git a/services/core/deploy/mcode/README.md b/services/core/deploy/mcode/README.md index 58cbbcea0..9272bd8a3 100644 --- a/services/core/deploy/mcode/README.md +++ b/services/core/deploy/mcode/README.md @@ -16,7 +16,7 @@ MiniMax Code accepts the `anthropic`, `responses` and `chat_completions` protoco ## Execution profiles -Each Session has its own native data directory, `$OAC_RUNTIME_HOME/runtime/mcode/state//` (`OAC_RUNTIME_HOME` defaults to `~/.oac`), holding the generated native configuration, instructions (`AGENTS.md`, at most 32 KiB), MCP configuration and history. The native process gets `MINIMAX_DATA_DIR`, `HOME` and `USERPROFILE` set to that directory on top of the daemon user's environment. +Each Session has its own native data directory, `$OAC_RUNTIME_HOME/runtime/mcode/state//` (`OAC_RUNTIME_HOME` defaults to `~/.oac`), holding the generated native configuration, instructions (`AGENTS.md`, at most 32 KiB), MCP configuration and history. The native process gets `MINIMAX_DATA_DIR`, `HOME` and `USERPROFILE` set to that directory on top of the daemon user's environment. This directory remains private to the compute instance: the adapter does not declare retained native history across compute replacement. | Profile | Where it runs | Native configuration | | --- | --- | --- | @@ -47,7 +47,7 @@ In the workspace profile, the frozen installation's Skills are linked into the S | Base | Digest-pinned `node:22.23.1-bookworm-slim` with `ca-certificates`, `bash`, `git`, `python3`, `python3-pip` and `ripgrep` | | Programs | `/usr/local/bin/oac-daemon` and the companion at `/opt/mcode-harness` (native CLI at `native/cli.js`, bridge at `bridge.mjs`) | | User | UID/GID 1000 with `HOME=/home/runtime` | -| Environment | `OAC_RUNTIME_HOME=/home/runtime/.oac`, `OAC_RUNTIME_MCODE_NODE`, `OAC_RUNTIME_MCODE_BIN`, `OAC_RUNTIME_MCODE_WORKSPACE_BRIDGE`, `OAC_RUNTIME_WORKSPACE=/environment/workspace`, `OAC_RUNTIME_INITIALIZATION_DIRECTORY=/environment/initialization`, `OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages` | +| Environment | `OAC_RUNTIME_HOME=/home/runtime/.oac`, `OAC_RUNTIME_MCODE_NODE`, `OAC_RUNTIME_MCODE_BIN`, `OAC_RUNTIME_MCODE_WORKSPACE_BRIDGE`, `OAC_RUNTIME_WORKSPACE=/environment/workspace`, `OAC_RUNTIME_INITIALIZATION_DIRECTORY=/environment/initialization`, `OAC_RUNTIME_PACKAGE_DIRECTORY=/environment/packages`, `OAC_RUNTIME_STATE_DIRECTORY=/environment/runtime-state` | | Entry point | `oac-daemon connect --profile default`, working directory `/environment/workspace` | The build runs the companion's `check.mjs` and the native `--version`. The combined Runtime image uses this image as its base. Sandboxes run it with the [Docker sandbox settings](../../../../docs/sandbox-provider.md#docker-adapter). diff --git a/services/core/deploy/runtime_profile.py b/services/core/deploy/runtime_profile.py new file mode 100644 index 000000000..cd4558dd7 --- /dev/null +++ b/services/core/deploy/runtime_profile.py @@ -0,0 +1,27 @@ +"""Preserve the Runtime's exported PATH in the pinned Debian login profile.""" +from pathlib import Path + +DEBIAN_PATH = '''if [ "$(id -u)" -eq 0 ]; then + PATH="/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin" +else + PATH="/usr/local/bin:/usr/bin:/bin:/usr/local/games:/usr/games" +fi +export PATH +''' + + +def preserve_runtime_path(profile: str) -> str: + if profile.count(DEBIAN_PATH) != 1: + raise ValueError("Pinned Debian profile PATH block changed") + # Bash creates an unexported PATH when its caller supplied none. Check the + # exported environment so that case still receives Debian's UID defaults. + replacement = ('# Preserve the Runtime tool environment across login shells.\n' + 'if ! /usr/bin/printenv PATH >/dev/null; then\n' + + ''.join(' ' + line for line in DEBIAN_PATH.splitlines(keepends=True)) + + 'fi\n') + return profile.replace(DEBIAN_PATH, replacement) + + +if __name__ == '__main__': + path = Path('/etc/profile') + path.write_text(preserve_runtime_path(path.read_text())) diff --git a/services/core/deploy/runtime_profile_test.py b/services/core/deploy/runtime_profile_test.py new file mode 100644 index 000000000..079f5936d --- /dev/null +++ b/services/core/deploy/runtime_profile_test.py @@ -0,0 +1,44 @@ +import os +from pathlib import Path +import subprocess +import tempfile +import unittest + +from runtime_profile import DEBIAN_PATH, preserve_runtime_path + + +class RuntimeProfileTest(unittest.TestCase): + def test_rejects_changed_or_duplicate_upstream_block(self): + for value in ('', DEBIAN_PATH.replace('/usr/games', '/changed'), DEBIAN_PATH * 2, + preserve_runtime_path(DEBIAN_PATH)): + with self.subTest(profile=value): + self.assertRaises(ValueError, preserve_runtime_path, value) + + def test_preserves_other_profile_content(self): + result = preserve_runtime_path('# before\n' + DEBIAN_PATH + '# after\n') + self.assertTrue(result.startswith('# before\n')) + self.assertTrue(result.endswith('# after\n')) + + @unittest.skipUnless(os.name == 'posix' and Path('/bin/bash').exists(), 'Bash profile') + def test_exported_unset_and_empty_path(self): + with tempfile.TemporaryDirectory() as directory: + original, fixed = Path(directory) / 'original', Path(directory) / 'fixed' + original.write_text(DEBIAN_PATH) + fixed.write_text(preserve_runtime_path(DEBIAN_PATH)) + def run(profile, path): + env = dict(os.environ) + env.pop('BASH_ENV', None) + if path is None: + env.pop('PATH', None) + else: + env['PATH'] = path + return subprocess.check_output(['/bin/bash', '--noprofile', '--norc', '-c', + '. "$1"; printf %s "$PATH"', 'profile-test', str(profile)], env=env, text=True) + custom = '/custom package/bin:/usr/bin:/bin' + self.assertEqual(run(fixed, custom), custom) + self.assertEqual(run(fixed, ''), '') + self.assertEqual(run(fixed, None), run(original, None)) + + +if __name__ == '__main__': + unittest.main() diff --git a/services/core/internal/db/queries/admin_session_archive.sql b/services/core/internal/db/queries/admin_session_archive.sql index 09b58224d..3562ee949 100644 --- a/services/core/internal/db/queries/admin_session_archive.sql +++ b/services/core/internal/db/queries/admin_session_archive.sql @@ -9,5 +9,5 @@ SELECT s.id AS session_id, e.id AS environment_id, ELSE 'active' END::text AS state FROM sessions s JOIN environments e ON e.session_id = s.id -LEFT JOIN runtime_allocations a ON a.environment_id = e.id +LEFT JOIN runtime_allocations a ON a.environment_id = e.id AND a.state <> 'released' WHERE s.tenant_id = $1 AND s.id = $2 AND s.deleted_at IS NULL; diff --git a/services/core/internal/db/queries/environments.sql b/services/core/internal/db/queries/environments.sql index 38fd0896f..2e122d372 100644 --- a/services/core/internal/db/queries/environments.sql +++ b/services/core/internal/db/queries/environments.sql @@ -4,11 +4,13 @@ VALUES ($1, $2, CASE WHEN EXISTS (SELECT 1 FROM initial_environment_files WHERE OR EXISTS (SELECT 1 FROM environment_setups WHERE session_id = $2) THEN 'pending' ELSE 'complete' END); -- name: GetEnvironment :one -SELECT sqlc.embed(e), s.tenant_id, (s.configuration->'environment')::jsonb AS configuration +SELECT sqlc.embed(e), s.tenant_id, (s.configuration->'environment')::jsonb AS configuration, + EXISTS (SELECT 1 FROM environment_workspaces w WHERE w.environment_id = e.id)::boolean AS external_workspace FROM environments e JOIN sessions s ON s.id = e.session_id WHERE s.tenant_id = $1 AND e.id = $2 AND s.deleted_at IS NULL; -- name: GetSessionEnvironment :one -SELECT sqlc.embed(e), s.tenant_id, (s.configuration->'environment')::jsonb AS configuration +SELECT sqlc.embed(e), s.tenant_id, (s.configuration->'environment')::jsonb AS configuration, + EXISTS (SELECT 1 FROM environment_workspaces w WHERE w.environment_id = e.id)::boolean AS external_workspace FROM environments e JOIN sessions s ON s.id = e.session_id WHERE s.tenant_id = $1 AND s.id = $2 AND s.deleted_at IS NULL; diff --git a/services/core/internal/db/queries/local_environment_devices.sql b/services/core/internal/db/queries/local_environment_devices.sql index 75d50bfc5..d7c756b0a 100644 --- a/services/core/internal/db/queries/local_environment_devices.sql +++ b/services/core/internal/db/queries/local_environment_devices.sql @@ -4,5 +4,27 @@ SELECT sqlc.arg(id), s.tenant_id, sqlc.arg(name), sqlc.arg(credential_hash), e.i FROM environments e JOIN sessions s ON s.id = e.session_id WHERE s.tenant_id = sqlc.arg(tenant_id) AND s.id = sqlc.arg(session_id) AND e.id = sqlc.arg(environment_id) AND s.deleted_at IS NULL AND s.configuration->'environment'->>'type' = 'openai_hosted' -ON CONFLICT (environment_id) DO NOTHING +AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state <> 'released') +AND NOT EXISTS (SELECT 1 FROM devices previous WHERE previous.environment_id = e.id AND (previous.revoked_at IS NULL OR previous.executor_key_id IS NOT NULL)) +ON CONFLICT (environment_id) WHERE revoked_at IS NULL OR executor_key_id IS NOT NULL DO NOTHING RETURNING id; + +-- name: BindHostedSessionDevice :one +INSERT INTO session_devices(session_id, device_id) +SELECT s.id, d.id FROM sessions s JOIN devices d ON d.tenant_id = s.tenant_id +JOIN environments e ON e.id = d.environment_id AND e.session_id = s.id +WHERE s.tenant_id = $1 AND s.id = $2 AND d.id = $3 AND d.revoked_at IS NULL + AND d.executor_key_id IS NULL AND s.deleted_at IS NULL + AND s.configuration->'environment'->>'type' = 'openai_hosted' +ON CONFLICT (session_id) DO UPDATE SET device_id = EXCLUDED.device_id +WHERE session_devices.device_id = EXCLUDED.device_id OR ( + EXISTS (SELECT 1 FROM devices previous + JOIN runtime_allocations a ON a.device_id = previous.id + WHERE previous.id = session_devices.device_id AND previous.revoked_at IS NOT NULL + AND previous.executor_key_id IS NULL AND a.state = 'released' + AND previous.environment_id = (SELECT environment_id FROM devices WHERE id = EXCLUDED.device_id)) + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a + WHERE a.environment_id = (SELECT environment_id FROM devices WHERE id = EXCLUDED.device_id) + AND a.state <> 'released') +) +RETURNING device_id; diff --git a/services/core/internal/db/queries/node_generations.sql b/services/core/internal/db/queries/node_generations.sql index 7b917e529..436268027 100644 --- a/services/core/internal/db/queries/node_generations.sql +++ b/services/core/internal/db/queries/node_generations.sql @@ -10,10 +10,10 @@ UNION ALL SELECT g.provider_kind,g.specification FROM runtime_deployment_generat LIMIT 1; -- name: UpsertNodeGenerationStatus :exec -INSERT INTO runtime_node_generation_status(node_id,generation,specification_digest,connection_id,owner_epoch,state,diagnostic) -VALUES($1,$2,$3,$4,$5,$6,$7) +INSERT INTO runtime_node_generation_status(node_id,generation,specification_digest,connection_id,owner_epoch,state,diagnostic,checkpoint) +VALUES($1,$2,$3,$4,$5,$6,$7,$8) ON CONFLICT(node_id,generation) DO UPDATE SET specification_digest=EXCLUDED.specification_digest,connection_id=EXCLUDED.connection_id, - owner_epoch=EXCLUDED.owner_epoch,state=EXCLUDED.state,diagnostic=EXCLUDED.diagnostic,observed_at=clock_timestamp(); + owner_epoch=EXCLUDED.owner_epoch,state=EXCLUDED.state,diagnostic=EXCLUDED.diagnostic,checkpoint=EXCLUDED.checkpoint,observed_at=clock_timestamp(); -- name: PromoteNodeServingGeneration :exec UPDATE runtime_nodes SET ready_generation=$2 WHERE id=$1 AND removed_at IS NULL @@ -33,3 +33,13 @@ SELECT EXISTS(SELECT 1 FROM runtime_node_generation_status g JOIN runtime_nodes -- name: DeleteNodeGenerationStatus :exec DELETE FROM runtime_node_generation_status WHERE node_id=$1 AND generation=$2; + +-- name: ListCheckpointGenerationNodes :many +SELECT n.id, g.checkpoint +FROM runtime_nodes n +JOIN runtime_node_generation_status g ON g.node_id=n.id +CROSS JOIN runtime_deployment d +WHERE g.generation=$1 AND n.installation_id=d.installation_id AND n.removed_at IS NULL +AND n.connection_id=g.connection_id AND n.connected_epoch=g.owner_epoch AND g.owner_epoch=d.owner_epoch +AND n.last_seen_at>clock_timestamp()-interval '45 seconds' AND g.state='ready' +ORDER BY n.id; diff --git a/services/core/internal/db/queries/runtime_allocations.sql b/services/core/internal/db/queries/runtime_allocations.sql index 4eeccef61..f724ca29d 100644 --- a/services/core/internal/db/queries/runtime_allocations.sql +++ b/services/core/internal/db/queries/runtime_allocations.sql @@ -2,12 +2,13 @@ INSERT INTO runtime_allocations (id, environment_id, device_id, provider_key, node_id, deployment_generation, compute_state) VALUES ($1, $2, $3, $4, $5, $6, jsonb_build_object('protocol_version', sqlc.arg(protocol_version)::text)) RETURNING *; --- name: GetRuntimeAllocation :one +-- name: GetLatestRuntimeAllocation :one SELECT sqlc.embed(a), e.session_id, s.tenant_id, s.deleted_at, (a.compute_phase NOT IN ('disabled', 'running') AND a.compute_retained_until IS NOT NULL AND a.compute_retained_until <= clock_timestamp())::boolean AS expired FROM runtime_allocations a JOIN environments e ON e.id = a.environment_id JOIN sessions s ON s.id = e.session_id -WHERE s.tenant_id = $1 AND a.environment_id = $2; +WHERE s.tenant_id = $1 AND a.environment_id = $2 +ORDER BY a.created_at DESC, a.id DESC LIMIT 1; -- name: ListRuntimeAllocations :many SELECT sqlc.embed(a), e.session_id, s.tenant_id, s.deleted_at, (a.compute_phase NOT IN ('disabled', 'running') AND a.compute_retained_until IS NOT NULL AND a.compute_retained_until <= clock_timestamp())::boolean AS expired @@ -21,7 +22,7 @@ ORDER BY a.id LIMIT 32; SELECT s.id, s.tenant_id FROM sessions s LEFT JOIN environments e ON e.session_id = s.id -LEFT JOIN runtime_allocations a ON a.environment_id = e.id +LEFT JOIN runtime_allocations a ON a.environment_id = e.id AND a.state <> 'released' WHERE s.id > $1 AND s.deleted_at IS NULL AND s.configuration->'environment'->>'type' = 'openai_hosted' @@ -45,3 +46,26 @@ WHERE id = $1 AND state <> 'released' RETURNING *; -- name: ReleaseRuntimeAllocation :one UPDATE runtime_allocations SET state = 'released', released_at = clock_timestamp() WHERE id = $1 AND state = 'cleanup_pending' AND create_settled RETURNING *; + +-- name: CanReplaceRuntimeAllocation :one +SELECT EXISTS ( + SELECT 1 FROM runtime_allocations a JOIN devices d ON d.id = a.device_id AND d.environment_id = a.environment_id + WHERE a.id = $1 AND a.state = 'released' AND a.create_settled + AND d.revoked_at IS NOT NULL AND d.executor_key_id IS NULL + AND NOT EXISTS (SELECT 1 FROM runtime_allocations current WHERE current.environment_id = a.environment_id AND current.state <> 'released') + AND NOT EXISTS (SELECT 1 FROM devices current WHERE current.environment_id = a.environment_id AND current.revoked_at IS NULL) +)::boolean; + +-- name: CanRetainRuntimeEnvironment :one +SELECT EXISTS ( + SELECT 1 FROM runtime_allocations a + JOIN devices d ON d.id = a.device_id AND d.environment_id = a.environment_id + JOIN environments e ON e.id = a.environment_id + JOIN sessions s ON s.id = e.session_id + JOIN environment_workspaces w ON w.environment_id = e.id AND w.state = 'ready' + WHERE a.id = $1 AND s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND e.initialization = 'complete' + AND s.configuration->'environment'->>'type' = 'openai_hosted' + AND d.supported_agent_kinds @> jsonb_build_array(jsonb_build_object( + 'kind', s.engine, 'available', true, 'capabilities', jsonb_build_object('retained_native_history', true))) +)::boolean; diff --git a/services/core/internal/db/queries/runtime_deployment.sql b/services/core/internal/db/queries/runtime_deployment.sql index 382d55726..332d6bbc3 100644 --- a/services/core/internal/db/queries/runtime_deployment.sql +++ b/services/core/internal/db/queries/runtime_deployment.sql @@ -4,10 +4,13 @@ SELECT * FROM runtime_deployment WHERE singleton = true FOR UPDATE; -- name: CountRuntimeDeploymentResources :one SELECT (SELECT count(*) FROM runtime_allocations WHERE state <> 'released')::bigint AS allocations, -(SELECT count(*) FROM environments e JOIN sessions s ON s.id = e.session_id - WHERE s.deleted_at IS NULL AND e.status = 'pending' +((SELECT count(*) FROM environments e JOIN sessions s ON s.id = e.session_id + WHERE s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND ((e.status = 'pending' AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id)) + OR EXISTS (SELECT 1 FROM runtime_placements p WHERE p.environment_id = e.id AND p.released_at IS NULL)) AND s.configuration->'environment'->>'type' = 'openai_hosted' - AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id))::bigint AS pending; + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state <> 'released'))::bigint + + (SELECT count(*) FROM runtime_reset_retained_environments)::bigint)::bigint AS pending; -- name: CountAddressBindings :one SELECT diff --git a/services/core/internal/db/queries/runtime_enrollment.sql b/services/core/internal/db/queries/runtime_enrollment.sql index 9fadd340c..0c68f0bcf 100644 --- a/services/core/internal/db/queries/runtime_enrollment.sql +++ b/services/core/internal/db/queries/runtime_enrollment.sql @@ -15,12 +15,12 @@ FOR SHARE OF c; -- name: EnrollRuntimeDevice :one INSERT INTO devices (id, tenant_id, name, environment_id, executor_key_id) VALUES (sqlc.arg(id), sqlc.arg(tenant_id), 'User-managed Runtime', sqlc.arg(environment_id), sqlc.arg(executor_key_id)) -ON CONFLICT (environment_id) DO UPDATE SET name = devices.name +ON CONFLICT (environment_id) WHERE revoked_at IS NULL OR executor_key_id IS NOT NULL DO UPDATE SET name = devices.name WHERE devices.executor_key_id = EXCLUDED.executor_key_id AND devices.revoked_at IS NULL RETURNING id, name, environment_id; -- name: TouchAuthenticatedDevice :execrows -UPDATE devices SET last_seen_at = clock_timestamp() +UPDATE devices SET last_seen_at = clock_timestamp(), supported_agent_kinds = sqlc.arg(supported_agent_kinds)::jsonb WHERE devices.id = sqlc.arg(id) AND EXISTS ( SELECT 1 FROM runtime_device_authority a WHERE a.id = devices.id AND a.credential_hash = sqlc.arg(credential_hash) diff --git a/services/core/internal/db/queries/runtime_lifecycle_nodes.sql b/services/core/internal/db/queries/runtime_lifecycle_nodes.sql index a30354cbd..d73c69e1f 100644 --- a/services/core/internal/db/queries/runtime_lifecycle_nodes.sql +++ b/services/core/internal/db/queries/runtime_lifecycle_nodes.sql @@ -19,18 +19,47 @@ SELECT e.id, s.tenant_id FROM environments e JOIN sessions s ON s.id=e.session_id LEFT JOIN runtime_placements p ON p.environment_id=e.id WHERE p.node_id IS NOT DISTINCT FROM sqlc.narg(node_id)::uuid + AND (p.environment_id IS NOT NULL OR (SELECT mode = 'direct' FROM runtime_deployment)) AND p.released_at IS NULL AND (SELECT reset_clear IS NULL FROM runtime_deployment) - AND e.id > sqlc.arg(after_id)::uuid AND s.deleted_at IS NULL AND e.status='pending' + AND e.id > sqlc.arg(after_id)::uuid AND s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') AND s.configuration->'environment'->>'type'='openai_hosted' - AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id=e.id) + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id=e.id AND a.state<>'released') ORDER BY e.id LIMIT 32; -- name: GetRuntimeLifecyclePlacement :one SELECT d.provider_kind, d.mode, a.id AS allocation_id, a.node_id AS allocation_node_id, - p.node_id AS placement_node_id, p.released_at + p.node_id AS placement_node_id, p.released_at, + CASE WHEN COALESCE(a.deployment_generation, p.deployment_generation, d.generation)=d.generation + THEN d.specification ELSE g.specification END::jsonb AS specification FROM environments e JOIN sessions s ON s.id=e.session_id CROSS JOIN runtime_deployment d -LEFT JOIN runtime_allocations a ON a.environment_id=e.id +LEFT JOIN runtime_allocations a ON a.environment_id=e.id AND a.state<>'released' LEFT JOIN runtime_placements p ON p.environment_id=e.id +LEFT JOIN runtime_deployment_generations g ON g.generation=COALESCE(a.deployment_generation, p.deployment_generation, d.generation) WHERE s.tenant_id=$1 AND e.id=$2; + +-- name: ListPlacementDemand :many +WITH demand AS ( +SELECT e.id, s.tenant_id, s.engine, (latest.id IS NOT NULL)::boolean AS retained, + CASE WHEN latest.id IS NULL THEN s.created_at + ELSE LEAST((SELECT min(r.created_at) FROM environment_input_reservations r WHERE r.session_id = s.id AND r.state = 'pending' AND r.deadline > clock_timestamp()), CASE WHEN latest.compute_wake_requested THEN latest.compute_activity_at END) END::timestamptz AS demanded_at +FROM environments e JOIN sessions s ON s.id = e.session_id +LEFT JOIN LATERAL (SELECT a.id, a.state, a.compute_wake_requested, a.compute_activity_at FROM runtime_allocations a WHERE a.environment_id = e.id ORDER BY a.created_at DESC, a.id DESC LIMIT 1) latest ON true +WHERE s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND s.configuration->'environment'->>'type' = 'openai_hosted' + AND (SELECT reset_clear IS NULL AND mode = 'nodes' FROM runtime_deployment) + AND NOT EXISTS (SELECT 1 FROM runtime_placements p WHERE p.environment_id = e.id AND p.released_at IS NULL) + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state <> 'released') + AND ( + (e.initialization IN ('pending','complete') AND latest.id IS NULL) + OR (e.initialization = 'complete' + AND EXISTS (SELECT 1 FROM environment_workspaces w WHERE w.environment_id = e.id AND w.state = 'ready') + AND EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state = 'released') + AND (latest.compute_wake_requested OR EXISTS (SELECT 1 FROM environment_input_reservations r WHERE r.session_id = s.id AND r.state = 'pending' AND r.deadline > clock_timestamp()))) + ) +) +SELECT id, tenant_id, engine, retained, demanded_at, COALESCE(sqlc.narg(until_time)::timestamptz, statement_timestamp())::timestamptz AS scan_until FROM demand +WHERE (demanded_at, id) > (sqlc.arg(after_time)::timestamptz, sqlc.arg(after_id)::uuid) +AND demanded_at <= COALESCE(sqlc.narg(until_time)::timestamptz, statement_timestamp()) +ORDER BY demanded_at, id LIMIT 32; diff --git a/services/core/internal/db/queries/runtime_nodes.sql b/services/core/internal/db/queries/runtime_nodes.sql index b6c45b2ce..44f68e596 100644 --- a/services/core/internal/db/queries/runtime_nodes.sql +++ b/services/core/internal/db/queries/runtime_nodes.sql @@ -14,12 +14,12 @@ SELECT n.*, (n.connection_id IS NOT NULL AND n.connected_epoch=d.owner_epoch AND EXISTS(SELECT 1 FROM runtime_node_generation_status g WHERE g.node_id=n.id AND g.generation=n.ready_generation AND g.connection_id=n.connection_id AND g.owner_epoch=d.owner_epoch AND g.state='ready')::boolean AS serving_ready, COALESCE((SELECT g.state FROM runtime_node_generation_status g WHERE g.node_id=n.id AND g.generation=d.generation AND g.connection_id=n.connection_id AND g.owner_epoch=d.owner_epoch),'')::text AS target_state, COALESCE((SELECT g.diagnostic FROM runtime_node_generation_status g WHERE g.node_id=n.id AND g.generation=d.generation AND g.connection_id=n.connection_id AND g.owner_epoch=d.owner_epoch),'')::text AS target_diagnostic, - (SELECT count(*) FROM runtime_placements p LEFT JOIN runtime_allocations a ON a.environment_id=p.environment_id WHERE p.node_id=n.id AND p.released_at IS NULL AND (a.id IS NULL OR a.compute_phase <> 'suspended'))::bigint AS active, + (SELECT count(*) FROM runtime_placements p LEFT JOIN runtime_allocations a ON a.environment_id=p.environment_id AND a.state<>'released' WHERE p.node_id=n.id AND p.released_at IS NULL AND (a.id IS NULL OR a.compute_phase <> 'suspended'))::bigint AS active, (SELECT count(*) FROM runtime_placements p WHERE p.node_id=n.id AND p.released_at IS NULL)::bigint AS retained, - (SELECT count(*) FROM runtime_placements p WHERE p.node_id=n.id AND p.released_at IS NULL AND NOT EXISTS(SELECT 1 FROM runtime_allocations a WHERE a.environment_id=p.environment_id))::bigint AS reserved, + (SELECT count(*) FROM runtime_placements p WHERE p.node_id=n.id AND p.released_at IS NULL AND NOT EXISTS(SELECT 1 FROM runtime_allocations a WHERE a.environment_id=p.environment_id AND a.state<>'released'))::bigint AS reserved, (SELECT count(*) FROM runtime_allocations a WHERE a.node_id=n.id AND a.state='cleanup_pending')::bigint AS cleanup_pending, (SELECT count(*) FROM runtime_allocations a WHERE a.node_id=n.id AND a.state='running' AND a.compute_phase IN('running','disabled'))::bigint AS running, - (SELECT count(*) FROM runtime_allocations a WHERE a.node_id=n.id AND a.state<>'released' AND a.compute_state->'snapshot' IS NOT NULL AND a.compute_state->'snapshot'<>'null'::jsonb)::bigint AS snapshots + (SELECT count(*) FROM runtime_allocations a WHERE a.node_id=n.id AND a.state<>'released' AND a.compute_state->'retained' IS NOT NULL AND a.compute_state->'retained'<>'null'::jsonb)::bigint AS snapshots FROM runtime_nodes n CROSS JOIN runtime_deployment d WHERE n.removed_at IS NULL AND n.installation_id=d.installation_id AND (sqlc.narg(node_id)::uuid IS NULL OR n.id=sqlc.narg(node_id)::uuid) ORDER BY n.id; @@ -63,7 +63,7 @@ INSERT INTO runtime_placements(environment_id,node_id,deployment_generation) VAL SELECT p.*, n.name, (EXISTS(SELECT 1 FROM runtime_node_generation_status g WHERE g.node_id=n.id AND g.generation=p.deployment_generation AND g.connection_id=n.connection_id AND g.owner_epoch=d.owner_epoch AND g.state='ready') AND n.connection_id IS NOT NULL AND n.connected_epoch=d.owner_epoch AND n.last_seen_at>clock_timestamp()-interval '45 seconds' AND n.removed_at IS NULL)::boolean AS available, COALESCE(a.observation_error,'')::text AS observation_error, COALESCE(a.state,'reserved')::text AS state, COALESCE(a.compute_phase,'disabled')::text AS compute_phase FROM runtime_placements p JOIN runtime_nodes n ON n.id=p.node_id CROSS JOIN runtime_deployment d -LEFT JOIN runtime_allocations a ON a.environment_id=p.environment_id WHERE p.environment_id=$1; +LEFT JOIN runtime_allocations a ON a.environment_id=p.environment_id AND a.state<>'released' WHERE p.environment_id=$1; -- name: ReleaseRuntimePlacement :exec UPDATE runtime_placements SET released_at=COALESCE(released_at,clock_timestamp()) WHERE environment_id=$1; @@ -71,7 +71,7 @@ UPDATE runtime_placements SET released_at=COALESCE(released_at,clock_timestamp() -- name: ReleaseUnallocatedRuntimePlacement :exec UPDATE runtime_placements SET released_at=COALESCE(released_at,clock_timestamp()) WHERE environment_id IN(SELECT id FROM environments WHERE session_id=$1) -AND NOT EXISTS(SELECT 1 FROM runtime_allocations a WHERE a.environment_id=runtime_placements.environment_id); +AND NOT EXISTS(SELECT 1 FROM runtime_allocations a WHERE a.environment_id=runtime_placements.environment_id AND a.state<>'released'); -- name: ListNodeRuntimeAllocations :many SELECT a.id,a.node_id,a.deployment_generation,a.observation_error,a.state,a.compute_phase,a.compute_phase_changed_at,e.initialization,a.created_at,a.environment_id,e.session_id,s.tenant_id @@ -84,3 +84,9 @@ SELECT id,$2,sqlc.arg(generation)::bigint FROM environments WHERE session_id=$1; -- name: SetRuntimeObservation :exec UPDATE runtime_allocations SET observation_error=$4 WHERE id=$1 AND compute_revision=$2 AND state=$3 AND state<>'released'; + +-- name: ReserveReleasedRuntimePlacement :execrows +UPDATE runtime_placements SET node_id = $2, deployment_generation = $3, + reserved_at = clock_timestamp(), released_at = NULL +WHERE runtime_placements.environment_id = $1 AND released_at IS NOT NULL + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = $1 AND a.state <> 'released'); diff --git a/services/core/internal/db/queries/runtime_suspension.sql b/services/core/internal/db/queries/runtime_suspension.sql index 2745f4e17..54a5523af 100644 --- a/services/core/internal/db/queries/runtime_suspension.sql +++ b/services/core/internal/db/queries/runtime_suspension.sql @@ -11,11 +11,11 @@ RETURNING *; -- name: TouchRuntimeActivity :exec UPDATE runtime_allocations a -SET compute_activity_at = clock_timestamp(), compute_wake_requested = true +SET compute_activity_at = CASE WHEN a.state = 'released' AND a.compute_wake_requested THEN a.compute_activity_at ELSE clock_timestamp() END, compute_wake_requested = true FROM environments e JOIN sessions s ON s.id = e.session_id WHERE a.environment_id = e.id AND s.tenant_id = sqlc.arg(tenant_id) AND e.id = sqlc.arg(environment_id) AND s.deleted_at IS NULL - AND a.state = 'running' AND a.compute_phase <> 'disabled'; + AND ((a.state = 'running' AND a.compute_phase <> 'disabled') OR a.id = sqlc.narg(retained_id)::uuid); -- name: ClearRuntimeWake :exec UPDATE runtime_allocations @@ -24,7 +24,7 @@ WHERE id = $1 AND compute_phase = 'running' AND compute_activity_at <= $2; -- name: GetRuntimeActivity :one SELECT clock_timestamp()::timestamptz AS observed_at, - GREATEST(a.compute_activity_at, + GREATEST(a.compute_activity_at, a.compute_phase_changed_at, (SELECT max(f.settled_at) FROM environment_file_writes f WHERE f.environment_id = e.id))::timestamptz AS last_activity, (EXISTS (SELECT 1 FROM turns t WHERE t.session_id = e.session_id AND t.status IN ('queued','in_progress','waiting')) OR EXISTS (SELECT 1 FROM subagent_turns t WHERE t.session_id = e.session_id AND t.status IN ('queued','in_progress','waiting')) @@ -37,7 +37,7 @@ WHERE a.id = $1; -- name: RuntimeComputeBlocksAdmission :one SELECT EXISTS ( SELECT 1 FROM runtime_allocations a JOIN environments e ON e.id = a.environment_id - WHERE e.session_id = $1 AND a.compute_phase NOT IN ('disabled', 'running') + WHERE e.session_id = $1 AND a.state <> 'released' AND a.compute_phase NOT IN ('disabled', 'running') )::boolean; -- name: RecordRuntimeTerminalActivity :exec @@ -49,14 +49,17 @@ WHERE a.environment_id = e.id AND e.session_id = $1 -- name: SessionHasRuntimeNode :one SELECT EXISTS ( SELECT 1 FROM runtime_allocations a JOIN environments e ON e.id = a.environment_id - WHERE e.session_id = $1 AND a.node_id IS NOT NULL + WHERE e.session_id = $1 AND a.state <> 'released' AND a.node_id IS NOT NULL )::boolean; -- name: HasIncompatibleRuntimeComputeState :one SELECT EXISTS ( SELECT 1 FROM runtime_allocations WHERE state <> 'released' - AND (compute_state->>'protocol_version') IS DISTINCT FROM sqlc.arg(protocol_version)::text + AND ((compute_state->>'protocol_version') IS DISTINCT FROM sqlc.arg(protocol_version)::text + OR compute_state ? 'snapshot' + OR (compute_state #> '{current,RestoredFrom}') ?| ARRAY['Digest', 'CheckpointID', 'CheckpointRoot'] + OR (compute_state #> '{target,RestoredFrom}') ?| ARRAY['Digest', 'CheckpointID', 'CheckpointRoot']) )::boolean; -- name: CountRuntimeComputeReservations :one @@ -66,3 +69,23 @@ WHERE provider_key = $1 AND state <> 'released' AND compute_phase <> 'suspended' -- name: CountRuntimeRetainedAllocations :one SELECT count(*) FROM runtime_allocations WHERE provider_key = $1 AND state <> 'released'; +-- name: MoveSuspendedRuntimeCompute :one +WITH moved AS ( + UPDATE runtime_placements p SET node_id=sqlc.arg(destination)::uuid + FROM runtime_allocations a + WHERE a.id=sqlc.arg(id) AND a.environment_id=p.environment_id + AND p.node_id=sqlc.arg(source)::uuid AND p.released_at IS NULL + AND p.deployment_generation=a.deployment_generation + AND a.node_id=p.node_id AND a.compute_revision=sqlc.arg(revision) + AND a.compute_phase='suspended' + AND ((a.state='running' AND sqlc.arg(phase)::text='restoring' + AND a.compute_retained_until>clock_timestamp() + AND EXISTS(SELECT 1 FROM environments e WHERE e.id=a.environment_id AND e.initialization='complete')) + OR (a.state='cleanup_pending' AND sqlc.arg(phase)::text='suspended')) + RETURNING a.id +) +UPDATE runtime_allocations a +SET node_id=sqlc.arg(destination)::uuid, compute_phase=sqlc.arg(phase), + compute_state=sqlc.arg(state)::jsonb, compute_revision=a.compute_revision+1, + compute_phase_changed_at=CASE WHEN a.compute_phase=sqlc.arg(phase)::text THEN a.compute_phase_changed_at ELSE clock_timestamp() END +FROM moved WHERE a.id=moved.id RETURNING a.*; diff --git a/services/core/internal/db/queries/runtime_suspension_pressure.sql b/services/core/internal/db/queries/runtime_suspension_pressure.sql new file mode 100644 index 000000000..4c47d4caa --- /dev/null +++ b/services/core/internal/db/queries/runtime_suspension_pressure.sql @@ -0,0 +1,14 @@ +-- name: ListRuntimeSuspensionInFlightNodes :many +SELECT DISTINCT a.node_id +FROM runtime_allocations a +WHERE a.node_id IS NOT NULL AND a.state <> 'released' + AND (a.compute_phase IN ('quiescing', 'suspending') + OR (a.compute_retained_until <= clock_timestamp() AND a.compute_phase <> 'disabled')); + +-- name: ListWaitingCheckpointRestores :many +SELECT DISTINCT a.node_id, a.deployment_generation, (a.compute_state->'retained'->'Compatibility')::jsonb AS checkpoint +FROM runtime_allocations a JOIN environments e ON e.id=a.environment_id JOIN sessions s ON s.id=e.session_id +WHERE a.node_id IS NOT NULL AND a.state='running' AND a.compute_phase='suspended' + AND a.compute_retained_until>clock_timestamp() AND s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND a.compute_state->'retained'->'Compatibility' IS NOT NULL + AND (a.compute_wake_requested OR EXISTS(SELECT 1 FROM environment_input_reservations r WHERE r.session_id=s.id AND r.state='pending' AND r.deadline>clock_timestamp())); diff --git a/services/core/internal/db/queries/sandbox_reset.sql b/services/core/internal/db/queries/sandbox_reset.sql index 536219525..5e5132e96 100644 --- a/services/core/internal/db/queries/sandbox_reset.sql +++ b/services/core/internal/db/queries/sandbox_reset.sql @@ -43,7 +43,9 @@ WHERE s.deleted_at IS NULL AND s.configuration->'environment'->>'type' = 'openai AND e.status NOT IN ('failed', 'expired') AND s.id > sqlc.arg(after_id)::uuid AND (EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state NOT IN ('released', 'cleanup_pending')) - OR (e.status = 'pending' AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id))) + OR (e.status = 'pending' AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id)) + OR EXISTS (SELECT 1 FROM runtime_placements p WHERE p.environment_id = e.id AND p.released_at IS NULL) + OR EXISTS (SELECT 1 FROM runtime_reset_retained_environments r WHERE r.environment_id = e.id)) AND (sqlc.arg(force)::boolean OR NOT ( EXISTS (SELECT 1 FROM turns t WHERE t.session_id = s.id AND t.status IN ('in_progress', 'waiting')) OR EXISTS (SELECT 1 FROM subagent_turns t WHERE t.session_id = s.id AND t.status IN ('in_progress', 'waiting')) @@ -69,9 +71,13 @@ held AS ( SELECT p.deployment_generation, p.node_id, s.id, e.id, true, false FROM environments e JOIN sessions s ON s.id = e.session_id LEFT JOIN runtime_placements p ON p.environment_id = e.id AND p.released_at IS NULL - WHERE s.deleted_at IS NULL AND e.status = 'pending' + WHERE s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND ((e.status = 'pending' AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id)) OR p.environment_id IS NOT NULL) AND s.configuration->'environment'->>'type' = 'openai_hosted' - AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id) + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state <> 'released') + UNION ALL + SELECT r.deployment_generation, NULL::uuid, r.session_id, r.environment_id, true, false + FROM runtime_reset_retained_environments r ), classified AS ( SELECT h.*, CASE WHEN h.cleanup THEN 'cleanup' WHEN EXISTS (SELECT 1 FROM turns t WHERE t.session_id = h.session_id AND t.status IN ('in_progress', 'waiting')) diff --git a/services/core/internal/db/queries/turns.sql b/services/core/internal/db/queries/turns.sql index 0caa8416c..4fab9fdd3 100644 --- a/services/core/internal/db/queries/turns.sql +++ b/services/core/internal/db/queries/turns.sql @@ -1,5 +1,5 @@ -- name: LockSession :one -SELECT id, deleted_at FROM sessions WHERE tenant_id = $1 AND id = $2 FOR UPDATE; +SELECT id, deleted_at, engine FROM sessions WHERE tenant_id = $1 AND id = $2 FOR UPDATE; -- name: GetActiveTurn :one SELECT * FROM turns WHERE session_id = $1 AND status IN ('queued', 'in_progress', 'waiting'); diff --git a/services/core/internal/db/sqlc/admin_session_archive.sql.go b/services/core/internal/db/sqlc/admin_session_archive.sql.go index 144700226..f494fadfd 100644 --- a/services/core/internal/db/sqlc/admin_session_archive.sql.go +++ b/services/core/internal/db/sqlc/admin_session_archive.sql.go @@ -22,7 +22,7 @@ SELECT s.id AS session_id, e.id AS environment_id, ELSE 'active' END::text AS state FROM sessions s JOIN environments e ON e.session_id = s.id -LEFT JOIN runtime_allocations a ON a.environment_id = e.id +LEFT JOIN runtime_allocations a ON a.environment_id = e.id AND a.state <> 'released' WHERE s.tenant_id = $1 AND s.id = $2 AND s.deleted_at IS NULL ` diff --git a/services/core/internal/db/sqlc/environments.sql.go b/services/core/internal/db/sqlc/environments.sql.go index f1c3c45e9..6a8097c94 100644 --- a/services/core/internal/db/sqlc/environments.sql.go +++ b/services/core/internal/db/sqlc/environments.sql.go @@ -28,7 +28,8 @@ func (q *Queries) CreateEnvironment(ctx context.Context, arg CreateEnvironmentPa } const getEnvironment = `-- name: GetEnvironment :one -SELECT e.id, e.session_id, e.status, e.created_at, e.failure_reason, e.failed_at, e.failure_detail, e.initialization, s.tenant_id, (s.configuration->'environment')::jsonb AS configuration +SELECT e.id, e.session_id, e.status, e.created_at, e.failure_reason, e.failed_at, e.failure_detail, e.initialization, s.tenant_id, (s.configuration->'environment')::jsonb AS configuration, + EXISTS (SELECT 1 FROM environment_workspaces w WHERE w.environment_id = e.id)::boolean AS external_workspace FROM environments e JOIN sessions s ON s.id = e.session_id WHERE s.tenant_id = $1 AND e.id = $2 AND s.deleted_at IS NULL ` @@ -39,9 +40,10 @@ type GetEnvironmentParams struct { } type GetEnvironmentRow struct { - Environment Environment `json:"environment"` - TenantID pgtype.UUID `json:"tenant_id"` - Configuration []byte `json:"configuration"` + Environment Environment `json:"environment"` + TenantID pgtype.UUID `json:"tenant_id"` + Configuration []byte `json:"configuration"` + ExternalWorkspace bool `json:"external_workspace"` } func (q *Queries) GetEnvironment(ctx context.Context, arg GetEnvironmentParams) (GetEnvironmentRow, error) { @@ -58,12 +60,14 @@ func (q *Queries) GetEnvironment(ctx context.Context, arg GetEnvironmentParams) &i.Environment.Initialization, &i.TenantID, &i.Configuration, + &i.ExternalWorkspace, ) return i, err } const getSessionEnvironment = `-- name: GetSessionEnvironment :one -SELECT e.id, e.session_id, e.status, e.created_at, e.failure_reason, e.failed_at, e.failure_detail, e.initialization, s.tenant_id, (s.configuration->'environment')::jsonb AS configuration +SELECT e.id, e.session_id, e.status, e.created_at, e.failure_reason, e.failed_at, e.failure_detail, e.initialization, s.tenant_id, (s.configuration->'environment')::jsonb AS configuration, + EXISTS (SELECT 1 FROM environment_workspaces w WHERE w.environment_id = e.id)::boolean AS external_workspace FROM environments e JOIN sessions s ON s.id = e.session_id WHERE s.tenant_id = $1 AND s.id = $2 AND s.deleted_at IS NULL ` @@ -74,9 +78,10 @@ type GetSessionEnvironmentParams struct { } type GetSessionEnvironmentRow struct { - Environment Environment `json:"environment"` - TenantID pgtype.UUID `json:"tenant_id"` - Configuration []byte `json:"configuration"` + Environment Environment `json:"environment"` + TenantID pgtype.UUID `json:"tenant_id"` + Configuration []byte `json:"configuration"` + ExternalWorkspace bool `json:"external_workspace"` } func (q *Queries) GetSessionEnvironment(ctx context.Context, arg GetSessionEnvironmentParams) (GetSessionEnvironmentRow, error) { @@ -93,6 +98,7 @@ func (q *Queries) GetSessionEnvironment(ctx context.Context, arg GetSessionEnvir &i.Environment.Initialization, &i.TenantID, &i.Configuration, + &i.ExternalWorkspace, ) return i, err } diff --git a/services/core/internal/db/sqlc/local_environment_devices.sql.go b/services/core/internal/db/sqlc/local_environment_devices.sql.go index 72c89655c..09774b54d 100644 --- a/services/core/internal/db/sqlc/local_environment_devices.sql.go +++ b/services/core/internal/db/sqlc/local_environment_devices.sql.go @@ -11,13 +11,49 @@ import ( "github.com/jackc/pgx/v5/pgtype" ) +const bindHostedSessionDevice = `-- name: BindHostedSessionDevice :one +INSERT INTO session_devices(session_id, device_id) +SELECT s.id, d.id FROM sessions s JOIN devices d ON d.tenant_id = s.tenant_id +JOIN environments e ON e.id = d.environment_id AND e.session_id = s.id +WHERE s.tenant_id = $1 AND s.id = $2 AND d.id = $3 AND d.revoked_at IS NULL + AND d.executor_key_id IS NULL AND s.deleted_at IS NULL + AND s.configuration->'environment'->>'type' = 'openai_hosted' +ON CONFLICT (session_id) DO UPDATE SET device_id = EXCLUDED.device_id +WHERE session_devices.device_id = EXCLUDED.device_id OR ( + EXISTS (SELECT 1 FROM devices previous + JOIN runtime_allocations a ON a.device_id = previous.id + WHERE previous.id = session_devices.device_id AND previous.revoked_at IS NOT NULL + AND previous.executor_key_id IS NULL AND a.state = 'released' + AND previous.environment_id = (SELECT environment_id FROM devices WHERE id = EXCLUDED.device_id)) + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a + WHERE a.environment_id = (SELECT environment_id FROM devices WHERE id = EXCLUDED.device_id) + AND a.state <> 'released') +) +RETURNING device_id +` + +type BindHostedSessionDeviceParams struct { + TenantID pgtype.UUID `json:"tenant_id"` + ID pgtype.UUID `json:"id"` + ID_2 pgtype.UUID `json:"id_2"` +} + +func (q *Queries) BindHostedSessionDevice(ctx context.Context, arg BindHostedSessionDeviceParams) (pgtype.UUID, error) { + row := q.db.QueryRow(ctx, bindHostedSessionDevice, arg.TenantID, arg.ID, arg.ID_2) + var device_id pgtype.UUID + err := row.Scan(&device_id) + return device_id, err +} + const createEnvironmentDevice = `-- name: CreateEnvironmentDevice :one INSERT INTO devices (id, tenant_id, name, credential_hash, environment_id) SELECT $1, s.tenant_id, $2, $3, e.id FROM environments e JOIN sessions s ON s.id = e.session_id WHERE s.tenant_id = $4 AND s.id = $5 AND e.id = $6 AND s.deleted_at IS NULL AND s.configuration->'environment'->>'type' = 'openai_hosted' -ON CONFLICT (environment_id) DO NOTHING +AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state <> 'released') +AND NOT EXISTS (SELECT 1 FROM devices previous WHERE previous.environment_id = e.id AND (previous.revoked_at IS NULL OR previous.executor_key_id IS NOT NULL)) +ON CONFLICT (environment_id) WHERE revoked_at IS NULL OR executor_key_id IS NOT NULL DO NOTHING RETURNING id ` diff --git a/services/core/internal/db/sqlc/models.go b/services/core/internal/db/sqlc/models.go index 70dacf734..d011c8983 100644 --- a/services/core/internal/db/sqlc/models.go +++ b/services/core/internal/db/sqlc/models.go @@ -64,6 +64,7 @@ type Device struct { EnvironmentID pgtype.UUID `json:"environment_id"` ExecutorKeyID pgtype.UUID `json:"executor_key_id"` ArchiveCancelTurnID pgtype.UUID `json:"archive_cancel_turn_id"` + SupportedAgentKinds []byte `json:"supported_agent_kinds"` } type Environment struct { @@ -341,6 +342,7 @@ type RuntimeNodeGenerationStatus struct { State string `json:"state"` Diagnostic string `json:"diagnostic"` ObservedAt pgtype.Timestamptz `json:"observed_at"` + Checkpoint []byte `json:"checkpoint"` } type RuntimePlacement struct { @@ -351,6 +353,13 @@ type RuntimePlacement struct { DeploymentGeneration pgtype.Int8 `json:"deployment_generation"` } +type RuntimeResetRetainedEnvironment struct { + EnvironmentID pgtype.UUID `json:"environment_id"` + SessionID pgtype.UUID `json:"session_id"` + TenantID pgtype.UUID `json:"tenant_id"` + DeploymentGeneration pgtype.Int8 `json:"deployment_generation"` +} + type Session struct { ID pgtype.UUID `json:"id"` TenantID pgtype.UUID `json:"tenant_id"` diff --git a/services/core/internal/db/sqlc/node_generations.sql.go b/services/core/internal/db/sqlc/node_generations.sql.go index ec512ca60..f52041660 100644 --- a/services/core/internal/db/sqlc/node_generations.sql.go +++ b/services/core/internal/db/sqlc/node_generations.sql.go @@ -43,6 +43,42 @@ func (q *Queries) GetNodeGenerationSpecification(ctx context.Context, generation return i, err } +const listCheckpointGenerationNodes = `-- name: ListCheckpointGenerationNodes :many +SELECT n.id, g.checkpoint +FROM runtime_nodes n +JOIN runtime_node_generation_status g ON g.node_id=n.id +CROSS JOIN runtime_deployment d +WHERE g.generation=$1 AND n.installation_id=d.installation_id AND n.removed_at IS NULL +AND n.connection_id=g.connection_id AND n.connected_epoch=g.owner_epoch AND g.owner_epoch=d.owner_epoch +AND n.last_seen_at>clock_timestamp()-interval '45 seconds' AND g.state='ready' +ORDER BY n.id +` + +type ListCheckpointGenerationNodesRow struct { + ID pgtype.UUID `json:"id"` + Checkpoint []byte `json:"checkpoint"` +} + +func (q *Queries) ListCheckpointGenerationNodes(ctx context.Context, generation int64) ([]ListCheckpointGenerationNodesRow, error) { + rows, err := q.db.Query(ctx, listCheckpointGenerationNodes, generation) + if err != nil { + return nil, err + } + defer rows.Close() + items := []ListCheckpointGenerationNodesRow{} + for rows.Next() { + var i ListCheckpointGenerationNodesRow + if err := rows.Scan(&i.ID, &i.Checkpoint); err != nil { + return nil, err + } + items = append(items, i) + } + if err := rows.Err(); err != nil { + return nil, err + } + return items, nil +} + const nodeGenerationKept = `-- name: NodeGenerationKept :one SELECT (EXISTS(SELECT 1 FROM runtime_deployment d WHERE d.generation=$1) OR EXISTS(SELECT 1 FROM runtime_nodes n WHERE n.id=$2 AND n.removed_at IS NULL AND n.ready_generation=$1) @@ -114,10 +150,10 @@ func (q *Queries) RefreshNodeServingReadiness(ctx context.Context, arg RefreshNo } const upsertNodeGenerationStatus = `-- name: UpsertNodeGenerationStatus :exec -INSERT INTO runtime_node_generation_status(node_id,generation,specification_digest,connection_id,owner_epoch,state,diagnostic) -VALUES($1,$2,$3,$4,$5,$6,$7) +INSERT INTO runtime_node_generation_status(node_id,generation,specification_digest,connection_id,owner_epoch,state,diagnostic,checkpoint) +VALUES($1,$2,$3,$4,$5,$6,$7,$8) ON CONFLICT(node_id,generation) DO UPDATE SET specification_digest=EXCLUDED.specification_digest,connection_id=EXCLUDED.connection_id, - owner_epoch=EXCLUDED.owner_epoch,state=EXCLUDED.state,diagnostic=EXCLUDED.diagnostic,observed_at=clock_timestamp() + owner_epoch=EXCLUDED.owner_epoch,state=EXCLUDED.state,diagnostic=EXCLUDED.diagnostic,checkpoint=EXCLUDED.checkpoint,observed_at=clock_timestamp() ` type UpsertNodeGenerationStatusParams struct { @@ -128,6 +164,7 @@ type UpsertNodeGenerationStatusParams struct { OwnerEpoch int64 `json:"owner_epoch"` State string `json:"state"` Diagnostic string `json:"diagnostic"` + Checkpoint []byte `json:"checkpoint"` } func (q *Queries) UpsertNodeGenerationStatus(ctx context.Context, arg UpsertNodeGenerationStatusParams) error { @@ -139,6 +176,7 @@ func (q *Queries) UpsertNodeGenerationStatus(ctx context.Context, arg UpsertNode arg.OwnerEpoch, arg.State, arg.Diagnostic, + arg.Checkpoint, ) return err } diff --git a/services/core/internal/db/sqlc/runtime_allocations.sql.go b/services/core/internal/db/sqlc/runtime_allocations.sql.go index 2eed749c1..d3ae628ad 100644 --- a/services/core/internal/db/sqlc/runtime_allocations.sql.go +++ b/services/core/internal/db/sqlc/runtime_allocations.sql.go @@ -11,6 +11,45 @@ import ( "github.com/jackc/pgx/v5/pgtype" ) +const canReplaceRuntimeAllocation = `-- name: CanReplaceRuntimeAllocation :one +SELECT EXISTS ( + SELECT 1 FROM runtime_allocations a JOIN devices d ON d.id = a.device_id AND d.environment_id = a.environment_id + WHERE a.id = $1 AND a.state = 'released' AND a.create_settled + AND d.revoked_at IS NOT NULL AND d.executor_key_id IS NULL + AND NOT EXISTS (SELECT 1 FROM runtime_allocations current WHERE current.environment_id = a.environment_id AND current.state <> 'released') + AND NOT EXISTS (SELECT 1 FROM devices current WHERE current.environment_id = a.environment_id AND current.revoked_at IS NULL) +)::boolean +` + +func (q *Queries) CanReplaceRuntimeAllocation(ctx context.Context, id pgtype.UUID) (bool, error) { + row := q.db.QueryRow(ctx, canReplaceRuntimeAllocation, id) + var column_1 bool + err := row.Scan(&column_1) + return column_1, err +} + +const canRetainRuntimeEnvironment = `-- name: CanRetainRuntimeEnvironment :one +SELECT EXISTS ( + SELECT 1 FROM runtime_allocations a + JOIN devices d ON d.id = a.device_id AND d.environment_id = a.environment_id + JOIN environments e ON e.id = a.environment_id + JOIN sessions s ON s.id = e.session_id + JOIN environment_workspaces w ON w.environment_id = e.id AND w.state = 'ready' + WHERE a.id = $1 AND s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND e.initialization = 'complete' + AND s.configuration->'environment'->>'type' = 'openai_hosted' + AND d.supported_agent_kinds @> jsonb_build_array(jsonb_build_object( + 'kind', s.engine, 'available', true, 'capabilities', jsonb_build_object('retained_native_history', true))) +)::boolean +` + +func (q *Queries) CanRetainRuntimeEnvironment(ctx context.Context, id pgtype.UUID) (bool, error) { + row := q.db.QueryRow(ctx, canRetainRuntimeEnvironment, id) + var column_1 bool + err := row.Scan(&column_1) + return column_1, err +} + const createRuntimeAllocation = `-- name: CreateRuntimeAllocation :one INSERT INTO runtime_allocations (id, environment_id, device_id, provider_key, node_id, deployment_generation, compute_state) VALUES ($1, $2, $3, $4, $5, $6, jsonb_build_object('protocol_version', $7::text)) RETURNING id, environment_id, device_id, provider_key, state, create_settled, created_at, released_at, compute_phase, compute_revision, compute_state, compute_activity_at, compute_wake_requested, compute_retained_until, node_id, observation_error, compute_phase_changed_at, deployment_generation @@ -60,20 +99,21 @@ func (q *Queries) CreateRuntimeAllocation(ctx context.Context, arg CreateRuntime return i, err } -const getRuntimeAllocation = `-- name: GetRuntimeAllocation :one +const getLatestRuntimeAllocation = `-- name: GetLatestRuntimeAllocation :one SELECT a.id, a.environment_id, a.device_id, a.provider_key, a.state, a.create_settled, a.created_at, a.released_at, a.compute_phase, a.compute_revision, a.compute_state, a.compute_activity_at, a.compute_wake_requested, a.compute_retained_until, a.node_id, a.observation_error, a.compute_phase_changed_at, a.deployment_generation, e.session_id, s.tenant_id, s.deleted_at, (a.compute_phase NOT IN ('disabled', 'running') AND a.compute_retained_until IS NOT NULL AND a.compute_retained_until <= clock_timestamp())::boolean AS expired FROM runtime_allocations a JOIN environments e ON e.id = a.environment_id JOIN sessions s ON s.id = e.session_id WHERE s.tenant_id = $1 AND a.environment_id = $2 +ORDER BY a.created_at DESC, a.id DESC LIMIT 1 ` -type GetRuntimeAllocationParams struct { +type GetLatestRuntimeAllocationParams struct { TenantID pgtype.UUID `json:"tenant_id"` EnvironmentID pgtype.UUID `json:"environment_id"` } -type GetRuntimeAllocationRow struct { +type GetLatestRuntimeAllocationRow struct { RuntimeAllocation RuntimeAllocation `json:"runtime_allocation"` SessionID pgtype.UUID `json:"session_id"` TenantID pgtype.UUID `json:"tenant_id"` @@ -81,9 +121,9 @@ type GetRuntimeAllocationRow struct { Expired bool `json:"expired"` } -func (q *Queries) GetRuntimeAllocation(ctx context.Context, arg GetRuntimeAllocationParams) (GetRuntimeAllocationRow, error) { - row := q.db.QueryRow(ctx, getRuntimeAllocation, arg.TenantID, arg.EnvironmentID) - var i GetRuntimeAllocationRow +func (q *Queries) GetLatestRuntimeAllocation(ctx context.Context, arg GetLatestRuntimeAllocationParams) (GetLatestRuntimeAllocationRow, error) { + row := q.db.QueryRow(ctx, getLatestRuntimeAllocation, arg.TenantID, arg.EnvironmentID) + var i GetLatestRuntimeAllocationRow err := row.Scan( &i.RuntimeAllocation.ID, &i.RuntimeAllocation.EnvironmentID, @@ -175,7 +215,7 @@ const listRuntimeObservationSessions = `-- name: ListRuntimeObservationSessions SELECT s.id, s.tenant_id FROM sessions s LEFT JOIN environments e ON e.session_id = s.id -LEFT JOIN runtime_allocations a ON a.environment_id = e.id +LEFT JOIN runtime_allocations a ON a.environment_id = e.id AND a.state <> 'released' WHERE s.id > $1 AND s.deleted_at IS NULL AND s.configuration->'environment'->>'type' = 'openai_hosted' diff --git a/services/core/internal/db/sqlc/runtime_deployment.sql.go b/services/core/internal/db/sqlc/runtime_deployment.sql.go index 56e51e53a..8f22aa200 100644 --- a/services/core/internal/db/sqlc/runtime_deployment.sql.go +++ b/services/core/internal/db/sqlc/runtime_deployment.sql.go @@ -34,10 +34,13 @@ func (q *Queries) CountAddressBindings(ctx context.Context, publicUrl string) (C const countRuntimeDeploymentResources = `-- name: CountRuntimeDeploymentResources :one SELECT (SELECT count(*) FROM runtime_allocations WHERE state <> 'released')::bigint AS allocations, -(SELECT count(*) FROM environments e JOIN sessions s ON s.id = e.session_id - WHERE s.deleted_at IS NULL AND e.status = 'pending' +((SELECT count(*) FROM environments e JOIN sessions s ON s.id = e.session_id + WHERE s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND ((e.status = 'pending' AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id)) + OR EXISTS (SELECT 1 FROM runtime_placements p WHERE p.environment_id = e.id AND p.released_at IS NULL)) AND s.configuration->'environment'->>'type' = 'openai_hosted' - AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id))::bigint AS pending + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state <> 'released'))::bigint + + (SELECT count(*) FROM runtime_reset_retained_environments)::bigint)::bigint AS pending ` type CountRuntimeDeploymentResourcesRow struct { diff --git a/services/core/internal/db/sqlc/runtime_enrollment.sql.go b/services/core/internal/db/sqlc/runtime_enrollment.sql.go index 5bc1b1d0c..a34dcf9c5 100644 --- a/services/core/internal/db/sqlc/runtime_enrollment.sql.go +++ b/services/core/internal/db/sqlc/runtime_enrollment.sql.go @@ -48,7 +48,7 @@ func (q *Queries) AuthorizeRuntimeEnrollment(ctx context.Context, arg AuthorizeR const enrollRuntimeDevice = `-- name: EnrollRuntimeDevice :one INSERT INTO devices (id, tenant_id, name, environment_id, executor_key_id) VALUES ($1, $2, 'User-managed Runtime', $3, $4) -ON CONFLICT (environment_id) DO UPDATE SET name = devices.name +ON CONFLICT (environment_id) WHERE revoked_at IS NULL OR executor_key_id IS NOT NULL DO UPDATE SET name = devices.name WHERE devices.executor_key_id = EXCLUDED.executor_key_id AND devices.revoked_at IS NULL RETURNING id, name, environment_id ` @@ -121,20 +121,21 @@ func (q *Queries) ListEnrolledRuntimeBindings(ctx context.Context) ([]ListEnroll } const touchAuthenticatedDevice = `-- name: TouchAuthenticatedDevice :execrows -UPDATE devices SET last_seen_at = clock_timestamp() -WHERE devices.id = $1 AND EXISTS ( +UPDATE devices SET last_seen_at = clock_timestamp(), supported_agent_kinds = $1::jsonb +WHERE devices.id = $2 AND EXISTS ( SELECT 1 FROM runtime_device_authority a - WHERE a.id = devices.id AND a.credential_hash = $2 + WHERE a.id = devices.id AND a.credential_hash = $3 ) ` type TouchAuthenticatedDeviceParams struct { - ID pgtype.UUID `json:"id"` - CredentialHash string `json:"credential_hash"` + SupportedAgentKinds []byte `json:"supported_agent_kinds"` + ID pgtype.UUID `json:"id"` + CredentialHash string `json:"credential_hash"` } func (q *Queries) TouchAuthenticatedDevice(ctx context.Context, arg TouchAuthenticatedDeviceParams) (int64, error) { - result, err := q.db.Exec(ctx, touchAuthenticatedDevice, arg.ID, arg.CredentialHash) + result, err := q.db.Exec(ctx, touchAuthenticatedDevice, arg.SupportedAgentKinds, arg.ID, arg.CredentialHash) if err != nil { return 0, err } diff --git a/services/core/internal/db/sqlc/runtime_lifecycle_nodes.sql.go b/services/core/internal/db/sqlc/runtime_lifecycle_nodes.sql.go index c1f443347..0a1cb8e65 100644 --- a/services/core/internal/db/sqlc/runtime_lifecycle_nodes.sql.go +++ b/services/core/internal/db/sqlc/runtime_lifecycle_nodes.sql.go @@ -13,11 +13,14 @@ import ( const getRuntimeLifecyclePlacement = `-- name: GetRuntimeLifecyclePlacement :one SELECT d.provider_kind, d.mode, a.id AS allocation_id, a.node_id AS allocation_node_id, - p.node_id AS placement_node_id, p.released_at + p.node_id AS placement_node_id, p.released_at, + CASE WHEN COALESCE(a.deployment_generation, p.deployment_generation, d.generation)=d.generation + THEN d.specification ELSE g.specification END::jsonb AS specification FROM environments e JOIN sessions s ON s.id=e.session_id CROSS JOIN runtime_deployment d -LEFT JOIN runtime_allocations a ON a.environment_id=e.id +LEFT JOIN runtime_allocations a ON a.environment_id=e.id AND a.state<>'released' LEFT JOIN runtime_placements p ON p.environment_id=e.id +LEFT JOIN runtime_deployment_generations g ON g.generation=COALESCE(a.deployment_generation, p.deployment_generation, d.generation) WHERE s.tenant_id=$1 AND e.id=$2 ` @@ -33,6 +36,7 @@ type GetRuntimeLifecyclePlacementRow struct { AllocationNodeID pgtype.UUID `json:"allocation_node_id"` PlacementNodeID pgtype.UUID `json:"placement_node_id"` ReleasedAt pgtype.Timestamptz `json:"released_at"` + Specification []byte `json:"specification"` } func (q *Queries) GetRuntimeLifecyclePlacement(ctx context.Context, arg GetRuntimeLifecyclePlacementParams) (GetRuntimeLifecyclePlacementRow, error) { @@ -45,10 +49,79 @@ func (q *Queries) GetRuntimeLifecyclePlacement(ctx context.Context, arg GetRunti &i.AllocationNodeID, &i.PlacementNodeID, &i.ReleasedAt, + &i.Specification, ) return i, err } +const listPlacementDemand = `-- name: ListPlacementDemand :many +WITH demand AS ( +SELECT e.id, s.tenant_id, s.engine, (latest.id IS NOT NULL)::boolean AS retained, + CASE WHEN latest.id IS NULL THEN s.created_at + ELSE LEAST((SELECT min(r.created_at) FROM environment_input_reservations r WHERE r.session_id = s.id AND r.state = 'pending' AND r.deadline > clock_timestamp()), CASE WHEN latest.compute_wake_requested THEN latest.compute_activity_at END) END::timestamptz AS demanded_at +FROM environments e JOIN sessions s ON s.id = e.session_id +LEFT JOIN LATERAL (SELECT a.id, a.state, a.compute_wake_requested, a.compute_activity_at FROM runtime_allocations a WHERE a.environment_id = e.id ORDER BY a.created_at DESC, a.id DESC LIMIT 1) latest ON true +WHERE s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND s.configuration->'environment'->>'type' = 'openai_hosted' + AND (SELECT reset_clear IS NULL AND mode = 'nodes' FROM runtime_deployment) + AND NOT EXISTS (SELECT 1 FROM runtime_placements p WHERE p.environment_id = e.id AND p.released_at IS NULL) + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state <> 'released') + AND ( + (e.initialization IN ('pending','complete') AND latest.id IS NULL) + OR (e.initialization = 'complete' + AND EXISTS (SELECT 1 FROM environment_workspaces w WHERE w.environment_id = e.id AND w.state = 'ready') + AND EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state = 'released') + AND (latest.compute_wake_requested OR EXISTS (SELECT 1 FROM environment_input_reservations r WHERE r.session_id = s.id AND r.state = 'pending' AND r.deadline > clock_timestamp()))) + ) +) +SELECT id, tenant_id, engine, retained, demanded_at, COALESCE($1::timestamptz, statement_timestamp())::timestamptz AS scan_until FROM demand +WHERE (demanded_at, id) > ($2::timestamptz, $3::uuid) +AND demanded_at <= COALESCE($1::timestamptz, statement_timestamp()) +ORDER BY demanded_at, id LIMIT 32 +` + +type ListPlacementDemandParams struct { + UntilTime pgtype.Timestamptz `json:"until_time"` + AfterTime pgtype.Timestamptz `json:"after_time"` + AfterID pgtype.UUID `json:"after_id"` +} + +type ListPlacementDemandRow struct { + ID pgtype.UUID `json:"id"` + TenantID pgtype.UUID `json:"tenant_id"` + Engine string `json:"engine"` + Retained bool `json:"retained"` + DemandedAt pgtype.Timestamptz `json:"demanded_at"` + ScanUntil pgtype.Timestamptz `json:"scan_until"` +} + +func (q *Queries) ListPlacementDemand(ctx context.Context, arg ListPlacementDemandParams) ([]ListPlacementDemandRow, error) { + rows, err := q.db.Query(ctx, listPlacementDemand, arg.UntilTime, arg.AfterTime, arg.AfterID) + if err != nil { + return nil, err + } + defer rows.Close() + items := []ListPlacementDemandRow{} + for rows.Next() { + var i ListPlacementDemandRow + if err := rows.Scan( + &i.ID, + &i.TenantID, + &i.Engine, + &i.Retained, + &i.DemandedAt, + &i.ScanUntil, + ); err != nil { + return nil, err + } + items = append(items, i) + } + if err := rows.Err(); err != nil { + return nil, err + } + return items, nil +} + const listRuntimeAllocationsForNode = `-- name: ListRuntimeAllocationsForNode :many SELECT a.id, a.environment_id, a.device_id, a.provider_key, a.state, a.create_settled, a.created_at, a.released_at, a.compute_phase, a.compute_revision, a.compute_state, a.compute_activity_at, a.compute_wake_requested, a.compute_retained_until, a.node_id, a.observation_error, a.compute_phase_changed_at, a.deployment_generation, e.session_id, s.tenant_id, s.deleted_at, (a.compute_phase NOT IN ('disabled', 'running') AND a.compute_retained_until IS NOT NULL AND a.compute_retained_until <= clock_timestamp())::boolean AS expired FROM runtime_allocations a @@ -148,11 +221,12 @@ SELECT e.id, s.tenant_id FROM environments e JOIN sessions s ON s.id=e.session_id LEFT JOIN runtime_placements p ON p.environment_id=e.id WHERE p.node_id IS NOT DISTINCT FROM $1::uuid + AND (p.environment_id IS NOT NULL OR (SELECT mode = 'direct' FROM runtime_deployment)) AND p.released_at IS NULL AND (SELECT reset_clear IS NULL FROM runtime_deployment) - AND e.id > $2::uuid AND s.deleted_at IS NULL AND e.status='pending' + AND e.id > $2::uuid AND s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') AND s.configuration->'environment'->>'type'='openai_hosted' - AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id=e.id) + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id=e.id AND a.state<>'released') ORDER BY e.id LIMIT 32 ` diff --git a/services/core/internal/db/sqlc/runtime_nodes.sql.go b/services/core/internal/db/sqlc/runtime_nodes.sql.go index f6bc5b7d7..d664a022d 100644 --- a/services/core/internal/db/sqlc/runtime_nodes.sql.go +++ b/services/core/internal/db/sqlc/runtime_nodes.sql.go @@ -203,7 +203,7 @@ const getRuntimePlacement = `-- name: GetRuntimePlacement :one SELECT p.environment_id, p.node_id, p.reserved_at, p.released_at, p.deployment_generation, n.name, (EXISTS(SELECT 1 FROM runtime_node_generation_status g WHERE g.node_id=n.id AND g.generation=p.deployment_generation AND g.connection_id=n.connection_id AND g.owner_epoch=d.owner_epoch AND g.state='ready') AND n.connection_id IS NOT NULL AND n.connected_epoch=d.owner_epoch AND n.last_seen_at>clock_timestamp()-interval '45 seconds' AND n.removed_at IS NULL)::boolean AS available, COALESCE(a.observation_error,'')::text AS observation_error, COALESCE(a.state,'reserved')::text AS state, COALESCE(a.compute_phase,'disabled')::text AS compute_phase FROM runtime_placements p JOIN runtime_nodes n ON n.id=p.node_id CROSS JOIN runtime_deployment d -LEFT JOIN runtime_allocations a ON a.environment_id=p.environment_id WHERE p.environment_id=$1 +LEFT JOIN runtime_allocations a ON a.environment_id=p.environment_id AND a.state<>'released' WHERE p.environment_id=$1 ` type GetRuntimePlacementRow struct { @@ -383,12 +383,12 @@ SELECT n.id, n.installation_id, n.name, n.backend_fingerprint, n.credential_sha2 EXISTS(SELECT 1 FROM runtime_node_generation_status g WHERE g.node_id=n.id AND g.generation=n.ready_generation AND g.connection_id=n.connection_id AND g.owner_epoch=d.owner_epoch AND g.state='ready')::boolean AS serving_ready, COALESCE((SELECT g.state FROM runtime_node_generation_status g WHERE g.node_id=n.id AND g.generation=d.generation AND g.connection_id=n.connection_id AND g.owner_epoch=d.owner_epoch),'')::text AS target_state, COALESCE((SELECT g.diagnostic FROM runtime_node_generation_status g WHERE g.node_id=n.id AND g.generation=d.generation AND g.connection_id=n.connection_id AND g.owner_epoch=d.owner_epoch),'')::text AS target_diagnostic, - (SELECT count(*) FROM runtime_placements p LEFT JOIN runtime_allocations a ON a.environment_id=p.environment_id WHERE p.node_id=n.id AND p.released_at IS NULL AND (a.id IS NULL OR a.compute_phase <> 'suspended'))::bigint AS active, + (SELECT count(*) FROM runtime_placements p LEFT JOIN runtime_allocations a ON a.environment_id=p.environment_id AND a.state<>'released' WHERE p.node_id=n.id AND p.released_at IS NULL AND (a.id IS NULL OR a.compute_phase <> 'suspended'))::bigint AS active, (SELECT count(*) FROM runtime_placements p WHERE p.node_id=n.id AND p.released_at IS NULL)::bigint AS retained, - (SELECT count(*) FROM runtime_placements p WHERE p.node_id=n.id AND p.released_at IS NULL AND NOT EXISTS(SELECT 1 FROM runtime_allocations a WHERE a.environment_id=p.environment_id))::bigint AS reserved, + (SELECT count(*) FROM runtime_placements p WHERE p.node_id=n.id AND p.released_at IS NULL AND NOT EXISTS(SELECT 1 FROM runtime_allocations a WHERE a.environment_id=p.environment_id AND a.state<>'released'))::bigint AS reserved, (SELECT count(*) FROM runtime_allocations a WHERE a.node_id=n.id AND a.state='cleanup_pending')::bigint AS cleanup_pending, (SELECT count(*) FROM runtime_allocations a WHERE a.node_id=n.id AND a.state='running' AND a.compute_phase IN('running','disabled'))::bigint AS running, - (SELECT count(*) FROM runtime_allocations a WHERE a.node_id=n.id AND a.state<>'released' AND a.compute_state->'snapshot' IS NOT NULL AND a.compute_state->'snapshot'<>'null'::jsonb)::bigint AS snapshots + (SELECT count(*) FROM runtime_allocations a WHERE a.node_id=n.id AND a.state<>'released' AND a.compute_state->'retained' IS NOT NULL AND a.compute_state->'retained'<>'null'::jsonb)::bigint AS snapshots FROM runtime_nodes n CROSS JOIN runtime_deployment d WHERE n.removed_at IS NULL AND n.installation_id=d.installation_id AND ($1::uuid IS NULL OR n.id=$1::uuid) ORDER BY n.id @@ -504,7 +504,7 @@ func (q *Queries) ReleaseRuntimePlacement(ctx context.Context, environmentID pgt const releaseUnallocatedRuntimePlacement = `-- name: ReleaseUnallocatedRuntimePlacement :exec UPDATE runtime_placements SET released_at=COALESCE(released_at,clock_timestamp()) WHERE environment_id IN(SELECT id FROM environments WHERE session_id=$1) -AND NOT EXISTS(SELECT 1 FROM runtime_allocations a WHERE a.environment_id=runtime_placements.environment_id) +AND NOT EXISTS(SELECT 1 FROM runtime_allocations a WHERE a.environment_id=runtime_placements.environment_id AND a.state<>'released') ` func (q *Queries) ReleaseUnallocatedRuntimePlacement(ctx context.Context, sessionID pgtype.UUID) error { @@ -521,6 +521,27 @@ func (q *Queries) RemoveRuntimeNode(ctx context.Context, id pgtype.UUID) error { return err } +const reserveReleasedRuntimePlacement = `-- name: ReserveReleasedRuntimePlacement :execrows +UPDATE runtime_placements SET node_id = $2, deployment_generation = $3, + reserved_at = clock_timestamp(), released_at = NULL +WHERE runtime_placements.environment_id = $1 AND released_at IS NOT NULL + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = $1 AND a.state <> 'released') +` + +type ReserveReleasedRuntimePlacementParams struct { + EnvironmentID pgtype.UUID `json:"environment_id"` + NodeID pgtype.UUID `json:"node_id"` + DeploymentGeneration pgtype.Int8 `json:"deployment_generation"` +} + +func (q *Queries) ReserveReleasedRuntimePlacement(ctx context.Context, arg ReserveReleasedRuntimePlacementParams) (int64, error) { + result, err := q.db.Exec(ctx, reserveReleasedRuntimePlacement, arg.EnvironmentID, arg.NodeID, arg.DeploymentGeneration) + if err != nil { + return 0, err + } + return result.RowsAffected(), nil +} + const setRuntimeObservation = `-- name: SetRuntimeObservation :exec UPDATE runtime_allocations SET observation_error=$4 WHERE id=$1 AND compute_revision=$2 AND state=$3 AND state<>'released' ` diff --git a/services/core/internal/db/sqlc/runtime_suspension.sql.go b/services/core/internal/db/sqlc/runtime_suspension.sql.go index ae029f567..44949ca26 100644 --- a/services/core/internal/db/sqlc/runtime_suspension.sql.go +++ b/services/core/internal/db/sqlc/runtime_suspension.sql.go @@ -53,7 +53,7 @@ func (q *Queries) CountRuntimeRetainedAllocations(ctx context.Context, providerK const getRuntimeActivity = `-- name: GetRuntimeActivity :one SELECT clock_timestamp()::timestamptz AS observed_at, - GREATEST(a.compute_activity_at, + GREATEST(a.compute_activity_at, a.compute_phase_changed_at, (SELECT max(f.settled_at) FROM environment_file_writes f WHERE f.environment_id = e.id))::timestamptz AS last_activity, (EXISTS (SELECT 1 FROM turns t WHERE t.session_id = e.session_id AND t.status IN ('queued','in_progress','waiting')) OR EXISTS (SELECT 1 FROM subagent_turns t WHERE t.session_id = e.session_id AND t.status IN ('queued','in_progress','waiting')) @@ -87,7 +87,10 @@ const hasIncompatibleRuntimeComputeState = `-- name: HasIncompatibleRuntimeCompu SELECT EXISTS ( SELECT 1 FROM runtime_allocations WHERE state <> 'released' - AND (compute_state->>'protocol_version') IS DISTINCT FROM $1::text + AND ((compute_state->>'protocol_version') IS DISTINCT FROM $1::text + OR compute_state ? 'snapshot' + OR (compute_state #> '{current,RestoredFrom}') ?| ARRAY['Digest', 'CheckpointID', 'CheckpointRoot'] + OR (compute_state #> '{target,RestoredFrom}') ?| ARRAY['Digest', 'CheckpointID', 'CheckpointRoot']) )::boolean ` @@ -98,6 +101,70 @@ func (q *Queries) HasIncompatibleRuntimeComputeState(ctx context.Context, protoc return column_1, err } +const moveSuspendedRuntimeCompute = `-- name: MoveSuspendedRuntimeCompute :one +WITH moved AS ( + UPDATE runtime_placements p SET node_id=$1::uuid + FROM runtime_allocations a + WHERE a.id=$4 AND a.environment_id=p.environment_id + AND p.node_id=$5::uuid AND p.released_at IS NULL + AND p.deployment_generation=a.deployment_generation + AND a.node_id=p.node_id AND a.compute_revision=$6 + AND a.compute_phase='suspended' + AND ((a.state='running' AND $2::text='restoring' + AND a.compute_retained_until>clock_timestamp() + AND EXISTS(SELECT 1 FROM environments e WHERE e.id=a.environment_id AND e.initialization='complete')) + OR (a.state='cleanup_pending' AND $2::text='suspended')) + RETURNING a.id +) +UPDATE runtime_allocations a +SET node_id=$1::uuid, compute_phase=$2, + compute_state=$3::jsonb, compute_revision=a.compute_revision+1, + compute_phase_changed_at=CASE WHEN a.compute_phase=$2::text THEN a.compute_phase_changed_at ELSE clock_timestamp() END +FROM moved WHERE a.id=moved.id RETURNING a.id, a.environment_id, a.device_id, a.provider_key, a.state, a.create_settled, a.created_at, a.released_at, a.compute_phase, a.compute_revision, a.compute_state, a.compute_activity_at, a.compute_wake_requested, a.compute_retained_until, a.node_id, a.observation_error, a.compute_phase_changed_at, a.deployment_generation +` + +type MoveSuspendedRuntimeComputeParams struct { + Destination pgtype.UUID `json:"destination"` + Phase string `json:"phase"` + State []byte `json:"state"` + ID pgtype.UUID `json:"id"` + Source pgtype.UUID `json:"source"` + Revision int64 `json:"revision"` +} + +func (q *Queries) MoveSuspendedRuntimeCompute(ctx context.Context, arg MoveSuspendedRuntimeComputeParams) (RuntimeAllocation, error) { + row := q.db.QueryRow(ctx, moveSuspendedRuntimeCompute, + arg.Destination, + arg.Phase, + arg.State, + arg.ID, + arg.Source, + arg.Revision, + ) + var i RuntimeAllocation + err := row.Scan( + &i.ID, + &i.EnvironmentID, + &i.DeviceID, + &i.ProviderKey, + &i.State, + &i.CreateSettled, + &i.CreatedAt, + &i.ReleasedAt, + &i.ComputePhase, + &i.ComputeRevision, + &i.ComputeState, + &i.ComputeActivityAt, + &i.ComputeWakeRequested, + &i.ComputeRetainedUntil, + &i.NodeID, + &i.ObservationError, + &i.ComputePhaseChangedAt, + &i.DeploymentGeneration, + ) + return i, err +} + const recordRuntimeTerminalActivity = `-- name: RecordRuntimeTerminalActivity :exec UPDATE runtime_allocations a SET compute_activity_at = clock_timestamp() FROM environments e @@ -113,7 +180,7 @@ func (q *Queries) RecordRuntimeTerminalActivity(ctx context.Context, sessionID p const runtimeComputeBlocksAdmission = `-- name: RuntimeComputeBlocksAdmission :one SELECT EXISTS ( SELECT 1 FROM runtime_allocations a JOIN environments e ON e.id = a.environment_id - WHERE e.session_id = $1 AND a.compute_phase NOT IN ('disabled', 'running') + WHERE e.session_id = $1 AND a.state <> 'released' AND a.compute_phase NOT IN ('disabled', 'running') )::boolean ` @@ -127,7 +194,7 @@ func (q *Queries) RuntimeComputeBlocksAdmission(ctx context.Context, sessionID p const sessionHasRuntimeNode = `-- name: SessionHasRuntimeNode :one SELECT EXISTS ( SELECT 1 FROM runtime_allocations a JOIN environments e ON e.id = a.environment_id - WHERE e.session_id = $1 AND a.node_id IS NOT NULL + WHERE e.session_id = $1 AND a.state <> 'released' AND a.node_id IS NOT NULL )::boolean ` @@ -192,19 +259,20 @@ func (q *Queries) SetRuntimeCompute(ctx context.Context, arg SetRuntimeComputePa const touchRuntimeActivity = `-- name: TouchRuntimeActivity :exec UPDATE runtime_allocations a -SET compute_activity_at = clock_timestamp(), compute_wake_requested = true +SET compute_activity_at = CASE WHEN a.state = 'released' AND a.compute_wake_requested THEN a.compute_activity_at ELSE clock_timestamp() END, compute_wake_requested = true FROM environments e JOIN sessions s ON s.id = e.session_id WHERE a.environment_id = e.id AND s.tenant_id = $1 AND e.id = $2 AND s.deleted_at IS NULL - AND a.state = 'running' AND a.compute_phase <> 'disabled' + AND ((a.state = 'running' AND a.compute_phase <> 'disabled') OR a.id = $3::uuid) ` type TouchRuntimeActivityParams struct { TenantID pgtype.UUID `json:"tenant_id"` EnvironmentID pgtype.UUID `json:"environment_id"` + RetainedID pgtype.UUID `json:"retained_id"` } func (q *Queries) TouchRuntimeActivity(ctx context.Context, arg TouchRuntimeActivityParams) error { - _, err := q.db.Exec(ctx, touchRuntimeActivity, arg.TenantID, arg.EnvironmentID) + _, err := q.db.Exec(ctx, touchRuntimeActivity, arg.TenantID, arg.EnvironmentID, arg.RetainedID) return err } diff --git a/services/core/internal/db/sqlc/runtime_suspension_pressure.sql.go b/services/core/internal/db/sqlc/runtime_suspension_pressure.sql.go new file mode 100644 index 000000000..86b051015 --- /dev/null +++ b/services/core/internal/db/sqlc/runtime_suspension_pressure.sql.go @@ -0,0 +1,75 @@ +// Code generated by sqlc. DO NOT EDIT. +// versions: +// sqlc v1.29.0 +// source: runtime_suspension_pressure.sql + +package sqlc + +import ( + "context" + + "github.com/jackc/pgx/v5/pgtype" +) + +const listRuntimeSuspensionInFlightNodes = `-- name: ListRuntimeSuspensionInFlightNodes :many +SELECT DISTINCT a.node_id +FROM runtime_allocations a +WHERE a.node_id IS NOT NULL AND a.state <> 'released' + AND (a.compute_phase IN ('quiescing', 'suspending') + OR (a.compute_retained_until <= clock_timestamp() AND a.compute_phase <> 'disabled')) +` + +func (q *Queries) ListRuntimeSuspensionInFlightNodes(ctx context.Context) ([]pgtype.UUID, error) { + rows, err := q.db.Query(ctx, listRuntimeSuspensionInFlightNodes) + if err != nil { + return nil, err + } + defer rows.Close() + items := []pgtype.UUID{} + for rows.Next() { + var node_id pgtype.UUID + if err := rows.Scan(&node_id); err != nil { + return nil, err + } + items = append(items, node_id) + } + if err := rows.Err(); err != nil { + return nil, err + } + return items, nil +} + +const listWaitingCheckpointRestores = `-- name: ListWaitingCheckpointRestores :many +SELECT DISTINCT a.node_id, a.deployment_generation, (a.compute_state->'retained'->'Compatibility')::jsonb AS checkpoint +FROM runtime_allocations a JOIN environments e ON e.id=a.environment_id JOIN sessions s ON s.id=e.session_id +WHERE a.node_id IS NOT NULL AND a.state='running' AND a.compute_phase='suspended' + AND a.compute_retained_until>clock_timestamp() AND s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND a.compute_state->'retained'->'Compatibility' IS NOT NULL + AND (a.compute_wake_requested OR EXISTS(SELECT 1 FROM environment_input_reservations r WHERE r.session_id=s.id AND r.state='pending' AND r.deadline>clock_timestamp())) +` + +type ListWaitingCheckpointRestoresRow struct { + NodeID pgtype.UUID `json:"node_id"` + DeploymentGeneration pgtype.Int8 `json:"deployment_generation"` + Checkpoint []byte `json:"checkpoint"` +} + +func (q *Queries) ListWaitingCheckpointRestores(ctx context.Context) ([]ListWaitingCheckpointRestoresRow, error) { + rows, err := q.db.Query(ctx, listWaitingCheckpointRestores) + if err != nil { + return nil, err + } + defer rows.Close() + items := []ListWaitingCheckpointRestoresRow{} + for rows.Next() { + var i ListWaitingCheckpointRestoresRow + if err := rows.Scan(&i.NodeID, &i.DeploymentGeneration, &i.Checkpoint); err != nil { + return nil, err + } + items = append(items, i) + } + if err := rows.Err(); err != nil { + return nil, err + } + return items, nil +} diff --git a/services/core/internal/db/sqlc/sandbox_reset.sql.go b/services/core/internal/db/sqlc/sandbox_reset.sql.go index 8b1d18f72..fcc589d64 100644 --- a/services/core/internal/db/sqlc/sandbox_reset.sql.go +++ b/services/core/internal/db/sqlc/sandbox_reset.sql.go @@ -64,9 +64,13 @@ held AS ( SELECT p.deployment_generation, p.node_id, s.id, e.id, true, false FROM environments e JOIN sessions s ON s.id = e.session_id LEFT JOIN runtime_placements p ON p.environment_id = e.id AND p.released_at IS NULL - WHERE s.deleted_at IS NULL AND e.status = 'pending' + WHERE s.deleted_at IS NULL AND e.status NOT IN ('failed','expired') + AND ((e.status = 'pending' AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id)) OR p.environment_id IS NOT NULL) AND s.configuration->'environment'->>'type' = 'openai_hosted' - AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id) + AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state <> 'released') + UNION ALL + SELECT r.deployment_generation, NULL::uuid, r.session_id, r.environment_id, true, false + FROM runtime_reset_retained_environments r ), classified AS ( SELECT h.deployment_generation, h.node_id, h.session_id, h.environment_id, h.pending, h.cleanup, CASE WHEN h.cleanup THEN 'cleanup' WHEN EXISTS (SELECT 1 FROM turns t WHERE t.session_id = h.session_id AND t.status IN ('in_progress', 'waiting')) @@ -166,7 +170,9 @@ WHERE s.deleted_at IS NULL AND s.configuration->'environment'->>'type' = 'openai AND e.status NOT IN ('failed', 'expired') AND s.id > $1::uuid AND (EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id AND a.state NOT IN ('released', 'cleanup_pending')) - OR (e.status = 'pending' AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id))) + OR (e.status = 'pending' AND NOT EXISTS (SELECT 1 FROM runtime_allocations a WHERE a.environment_id = e.id)) + OR EXISTS (SELECT 1 FROM runtime_placements p WHERE p.environment_id = e.id AND p.released_at IS NULL) + OR EXISTS (SELECT 1 FROM runtime_reset_retained_environments r WHERE r.environment_id = e.id)) AND ($2::boolean OR NOT ( EXISTS (SELECT 1 FROM turns t WHERE t.session_id = s.id AND t.status IN ('in_progress', 'waiting')) OR EXISTS (SELECT 1 FROM subagent_turns t WHERE t.session_id = s.id AND t.status IN ('in_progress', 'waiting')) diff --git a/services/core/internal/db/sqlc/turns.sql.go b/services/core/internal/db/sqlc/turns.sql.go index b3093ccbc..0100959ee 100644 --- a/services/core/internal/db/sqlc/turns.sql.go +++ b/services/core/internal/db/sqlc/turns.sql.go @@ -235,7 +235,7 @@ func (q *Queries) ListTurnInputs(ctx context.Context, arg ListTurnInputsParams) } const lockSession = `-- name: LockSession :one -SELECT id, deleted_at FROM sessions WHERE tenant_id = $1 AND id = $2 FOR UPDATE +SELECT id, deleted_at, engine FROM sessions WHERE tenant_id = $1 AND id = $2 FOR UPDATE ` type LockSessionParams struct { @@ -246,12 +246,13 @@ type LockSessionParams struct { type LockSessionRow struct { ID pgtype.UUID `json:"id"` DeletedAt pgtype.Timestamptz `json:"deleted_at"` + Engine string `json:"engine"` } func (q *Queries) LockSession(ctx context.Context, arg LockSessionParams) (LockSessionRow, error) { row := q.db.QueryRow(ctx, lockSession, arg.TenantID, arg.ID) var i LockSessionRow - err := row.Scan(&i.ID, &i.DeletedAt) + err := row.Scan(&i.ID, &i.DeletedAt, &i.Engine) return i, err } diff --git a/services/core/internal/deployment/allocation.go b/services/core/internal/deployment/allocation.go index 820c398f2..50909468e 100644 --- a/services/core/internal/deployment/allocation.go +++ b/services/core/internal/deployment/allocation.go @@ -112,10 +112,31 @@ type ObservationSessionPage struct { // an existing one. type UnallocatedEnvironment struct{ ID, TenantID string } +// PlacementDemand is committed work awaiting a node reservation. Demand time +// belongs to Session creation or its oldest pending input, not to a retry. +type PlacementDemand struct { + UnallocatedEnvironment + Engine string + Retained bool + At time.Time +} + +// PlacementDemandCursor continues the bounded, oldest-demand-first scan. +// An empty EnvironmentID ends a page scan; Until may still describe a +// nonempty final page for callers that yield before processing every row. +type PlacementDemandCursor struct { + // Until fixes the database-clock horizon of one sweep, so new arrivals cannot prevent wraparound. + Until time.Time + At time.Time + EnvironmentID string +} + // LifecyclePlacement is what routing a hosted Environment to its node // lifecycle reads: the deployment's provider and mode, the Environment's // allocation and the node placement its Session reserved. type LifecyclePlacement struct { + // Specification belongs to the current allocation or reserved generation. + Specification json.RawMessage Provider, Mode string // AllocationID and AllocationNodeID are empty without an allocation or // its node. diff --git a/services/core/internal/deployment/allocations.go b/services/core/internal/deployment/allocations.go index 5c6630cd8..e26fcd634 100644 --- a/services/core/internal/deployment/allocations.go +++ b/services/core/internal/deployment/allocations.go @@ -46,7 +46,7 @@ func (e *ExecutionOperations) ReserveAllocation(ctx context.Context, key Allocat if err != nil { return err } - if found { + if found && existing.State != "released" { if existing.ProviderKey != installation { return ErrAllocationConflict } @@ -54,6 +54,15 @@ func (e *ExecutionOperations) ReserveAllocation(ctx context.Context, key Allocat result = existing return nil } + if found { + allowed, err := tx.CanReplaceAllocation() + if err != nil { + return err + } + if !allowed { + return ErrAllocationConflict + } + } // Existing receipts replay first, so retries and cleanup continue // while admission is closed. d, err := tx.LockDeployment() @@ -177,7 +186,19 @@ func (e *ExecutionOperations) cleanup(ctx context.Context, owner Allocation, abs if revoked.SessionDeleted { err = sessions.CancelWork(ctx, tx) } else { - err = sessions.TerminateEnvironment(ctx, tx, revoked.Expired, sessions.ProvisioningFailureReason, nil) + retain := false + if revoked.Expired { + retain, err = tx.CanRetainEnvironment(revoked) + if err != nil { + return err + } + } + // Compute retention ends independently of the Session only when its + // initialized filesystem and native history can continue on new compute. + // Explicit archive/reset makes this locked eligibility check false. + if !retain { + err = sessions.TerminateEnvironment(ctx, tx, revoked.Expired, sessions.ProvisioningFailureReason, nil) + } } if err != nil { return err @@ -202,10 +223,10 @@ func (e *ExecutionOperations) cleanup(ctx context.Context, owner Allocation, abs // SetCompute commits a compute phase before its external effects. The owner's // revision and the Session lock fence a stale lifecycle observation. Moving -// running compute to quiescing requires the idle timeout, which the -// transaction checks again against the activity it reads. +// running compute to quiescing requires idle compute and either the idle +// timeout or capacity pressure, rechecked in the transaction. func (e *ExecutionOperations) SetCompute(ctx context.Context, owner Allocation, phase string, state json.RawMessage, retainedUntil *time.Time, idleTimeout time.Duration) (Allocation, error) { - if !computeTransition(owner.ComputePhase, phase) || !json.Valid(state) || (owner.ComputePhase == "running" && phase == "quiescing" && idleTimeout <= 0) { + if owner.ComputePhase == "suspended" && phase == "restoring" || !computeTransition(owner.ComputePhase, phase) || !json.Valid(state) || (owner.ComputePhase == "running" && phase == "quiescing" && idleTimeout <= 0) { return Allocation{}, ErrInvalidInput } if phase != "running" && (retainedUntil == nil || retainedUntil.IsZero()) { @@ -227,24 +248,98 @@ func (e *ExecutionOperations) SetCompute(ctx context.Context, owner Allocation, if activity.Busy || activity.WakeRequested { return Allocation{}, ErrAllocationConflict } - if phase == "quiescing" && (!activity.ReadyToSuspend(idleTimeout) || current.ComputeActivityAt.After(owner.ComputeActivityAt)) { - return Allocation{}, ErrAllocationConflict - } - } - // A node-backed restore reserves capacity on its original node. - if current.ComputePhase == "suspended" && phase == "restoring" && current.NodeID != "" { - restore, err := tx.LoadRestore(current) - if err != nil { - return Allocation{}, err - } - if err := placement.CheckRestore(restore); err != nil { - return Allocation{}, err + if phase == "quiescing" { + if current.ComputeActivityAt.After(owner.ComputeActivityAt) { + return Allocation{}, ErrAllocationConflict + } + if !activity.ReadyToSuspend(idleTimeout) { + // A short quiet period also backs off a rejected daemon quiesce. + if !activity.ReadyToSuspend(15*time.Second) || current.NodeID == "" { + return Allocation{}, ErrNotIdle + } + allowed, err := e.canSuspendForDemand(tx, current) + if err != nil { + return Allocation{}, err + } + if !allowed { + return Allocation{}, ErrNotIdle + } + } } } return tx.SetCompute(current, ComputeChange{Phase: phase, State: state, RetainedUntil: retainedUntil}) }) } +// BeginRestore reserves a compatible destination before any native restore. +// The caller supplies a fresh restore intent without a target; the destination +// adapter derives the target after this commit. Snapshot portability proves +// that publication and source-local cleanup settled before suspension. +func (e *ExecutionOperations) BeginRestore(ctx context.Context, owner Allocation, compatibility sandbox.CheckpointCompatibility, state json.RawMessage) (Allocation, error) { + var object map[string]json.RawMessage + if (owner.NodeID != "" && compatibility.Validate() != nil) || json.Unmarshal(state, &object) != nil || object == nil { + return Allocation{}, ErrInvalidInput + } + return e.change(ctx, owner, true, func(tx AllocationTx, current Allocation) (Allocation, error) { + if current.State != "running" || current.ComputePhase != "suspended" || current.ComputeRevision != owner.ComputeRevision || current.Expired || current.ComputeRetainedUntil == nil { + return Allocation{}, ErrAllocationConflict + } + activity, err := tx.LoadActivity(current) + if err != nil { + return Allocation{}, err + } + if !activity.Busy && !activity.WakeRequested { + return Allocation{}, ErrAllocationConflict + } + if current.NodeID == "" { + return tx.SetCompute(current, ComputeChange{Phase: "restoring", State: state, RetainedUntil: current.ComputeRetainedUntil}) + } + d, nodes, err := tx.LoadCheckpointPlacement(current) + if err != nil { + return Allocation{}, err + } + if d.Resetting { + return Allocation{}, placement.ErrResetAdmission + } + if d.InstallationID != current.ProviderKey || d.Mode != string(sandbox.DeploymentNodes) { + return Allocation{}, placement.ErrNodeUnavailable + } + node, err := e.service.rules.ChooseCheckpoint(nodes, current.NodeID, compatibility, false) + if err != nil { + return Allocation{}, err + } + return tx.MoveSuspended(current, node, ComputeChange{Phase: "restoring", State: state, RetainedUntil: current.ComputeRetainedUntil}) + }) +} + +// RelocateCleanup durably assigns a portable suspended artifact to a node that +// can delete it. An unknown restore remains pinned to its existing target. +func (e *ExecutionOperations) RelocateCleanup(ctx context.Context, owner Allocation, compatibility sandbox.CheckpointCompatibility) (Allocation, error) { + if owner.NodeID == "" || compatibility.Validate() != nil { + return Allocation{}, ErrInvalidInput + } + return e.change(ctx, owner, false, func(tx AllocationTx, current Allocation) (Allocation, error) { + if current.State != "cleanup_pending" || current.ComputePhase != "suspended" || current.ComputeRevision != owner.ComputeRevision { + return Allocation{}, ErrAllocationConflict + } + d, nodes, err := tx.LoadCheckpointPlacement(current) + if err != nil { + return Allocation{}, err + } + if d.InstallationID != current.ProviderKey || d.Mode != string(sandbox.DeploymentNodes) { + return Allocation{}, placement.ErrNodeUnavailable + } + node, err := e.service.rules.ChooseCheckpoint(nodes, current.NodeID, compatibility, true) + if err != nil { + return Allocation{}, err + } + if node == current.NodeID { + return current, nil + } + return tx.MoveSuspended(current, node, ComputeChange{Phase: current.ComputePhase, State: current.ComputeState, RetainedUntil: current.ComputeRetainedUntil}) + }) +} + // RecordObservation records a diagnostic for the observed compute revision // and state of a node-backed allocation. It never releases ownership, changes // public readiness or authorizes replacement. diff --git a/services/core/internal/deployment/allocations_test.go b/services/core/internal/deployment/allocations_test.go index dcbfa2a4e..22d6ea3db 100644 --- a/services/core/internal/deployment/allocations_test.go +++ b/services/core/internal/deployment/allocations_test.go @@ -6,6 +6,8 @@ import ( "encoding/hex" "encoding/json" "errors" + "fmt" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "slices" "testing" "time" @@ -13,6 +15,7 @@ import ( "github.com/google/uuid" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" ) @@ -69,6 +72,19 @@ func (f *fakeReservationTx) LoadReserved() (placement.Reserved, error) { return f.loadReserved() } +func (f *fakeReservationTx) CanReplaceAllocation() (bool, error) { + unexpected(f.t, "CanReplaceAllocation") + return false, nil +} +func (f *fakeReservationTx) LoadNodes() ([]placement.Node, error) { + unexpected(f.t, "LoadNodes") + return nil, nil +} +func (f *fakeReservationTx) ReservePlacement(placement.Placement) error { + unexpected(f.t, "ReservePlacement") + return nil +} + func (f *fakeReservationTx) InsertAllocation(allocation NewAllocation) (Allocation, error) { if f.insertAllocation == nil { unexpected(f.t, "InsertAllocation") @@ -77,11 +93,15 @@ func (f *fakeReservationTx) InsertAllocation(allocation NewAllocation) (Allocati } type fakeAllocationTx struct { - t testing.TB - loadAllocation func() (Allocation, error) - loadSessionDevice func() (SessionDevice, bool, error) - settleCreation func(Allocation) (Allocation, error) - release func(Allocation) (Allocation, error) + canRetainEnvironment func(Allocation) (bool, error) + t testing.TB + loadActivity func(Allocation) (Activity, error) + loadSuspensionDemand func(Allocation) (SuspensionDemand, error) + setCompute func(Allocation, ComputeChange) (Allocation, error) + loadAllocation func() (Allocation, error) + loadSessionDevice func() (SessionDevice, bool, error) + settleCreation func(Allocation) (Allocation, error) + release func(Allocation) (Allocation, error) } func (f *fakeAllocationTx) LoadAllocation() (Allocation, error) { @@ -98,14 +118,37 @@ func (f *fakeAllocationTx) LoadSessionDevice() (SessionDevice, bool, error) { return f.loadSessionDevice() } -func (f *fakeAllocationTx) LoadActivity(Allocation) (Activity, error) { - unexpected(f.t, "LoadActivity") - return Activity{}, nil +func (f *fakeAllocationTx) LoadActivity(current Allocation) (Activity, error) { + if f.loadActivity == nil { + unexpected(f.t, "LoadActivity") + } + return f.loadActivity(current) +} + +func (f *fakeAllocationTx) LoadSuspensionDemand(current Allocation) (SuspensionDemand, error) { + if f.loadSuspensionDemand == nil { + unexpected(f.t, "LoadSuspensionDemand") + } + return f.loadSuspensionDemand(current) +} + +func (f *fakeAllocationTx) PlacementDemand(PlacementDemandCursor) ([]PlacementDemand, PlacementDemandCursor, error) { + unexpected(f.t, "PlacementDemand") + return nil, PlacementDemandCursor{}, nil } -func (f *fakeAllocationTx) LoadRestore(Allocation) (placement.Restore, error) { - unexpected(f.t, "LoadRestore") - return placement.Restore{}, nil +func (f *fakeAllocationTx) LoadGenerationSpecification(uint64) (GenerationSpecification, error) { + unexpected(f.t, "LoadGenerationSpecification") + return GenerationSpecification{}, nil +} + +func (f *fakeAllocationTx) LoadCheckpointPlacement(Allocation) (placement.Deployment, []placement.CheckpointNode, error) { + unexpected(f.t, "LoadCheckpointPlacement") + return placement.Deployment{}, nil, nil +} +func (f *fakeAllocationTx) MoveSuspended(Allocation, string, ComputeChange) (Allocation, error) { + unexpected(f.t, "MoveSuspended") + return Allocation{}, nil } func (f *fakeAllocationTx) ObserveRunning(Allocation) (Allocation, error) { @@ -127,9 +170,11 @@ func (f *fakeAllocationTx) Release(current Allocation) (Allocation, error) { return f.release(current) } -func (f *fakeAllocationTx) SetCompute(Allocation, ComputeChange) (Allocation, error) { - unexpected(f.t, "SetCompute") - return Allocation{}, nil +func (f *fakeAllocationTx) SetCompute(current Allocation, change ComputeChange) (Allocation, error) { + if f.setCompute == nil { + unexpected(f.t, "SetCompute") + } + return f.setCompute(current, change) } func (f *fakeAllocationTx) RecordObservation(Allocation, string) error { @@ -141,6 +186,7 @@ func (f *fakeAllocationTx) RecordObservation(Allocation, string) error { // methods cover a Session without an active Turn, pending input or input // activity change. type fakeCleanupTx struct { + canRetainEnvironment func(Allocation) (bool, error) *fakeAllocationTx revokeDevice func(Allocation) error requestCleanup func(Allocation) (Allocation, error) @@ -152,6 +198,13 @@ type fakeCleanupTx struct { cancelPendingInput func() error } +func (f *fakeCleanupTx) CanRetainEnvironment(current Allocation) (bool, error) { + if f.canRetainEnvironment == nil { + return false, nil + } + return f.canRetainEnvironment(current) +} + func (f *fakeCleanupTx) RevokeDevice(current Allocation) error { if f.revokeDevice == nil { unexpected(f.t, "RevokeDevice") @@ -258,7 +311,7 @@ func allocationOperations(t *testing.T, reservation *fakeReservationTx, locked s return apply(allocation) } } - result, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, testPublicURL), storage) + result, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, testPublicURL), storage, engine.Catalog{}) if err != nil { t.Fatal(err) } @@ -493,7 +546,7 @@ func TestCleanupRevokesSettlesTheSessionThenReleases(t *testing.T) { storage := &fakeExecutionStorage{t: t, withAllocationCleanup: func(_ context.Context, _ AllocationKey, apply func(AllocationCleanupTx) error) error { return apply(tx) }} - operations, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, testPublicURL), storage) + operations, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, testPublicURL), storage, engine.Catalog{}) if err != nil { t.Fatal(err) } @@ -508,3 +561,140 @@ func TestCleanupRevokesSettlesTheSessionThenReleases(t *testing.T) { }) } } + +func TestExpiredQualifiedFilesystemCleanupPreservesSessionWork(t *testing.T) { + owner := Allocation{ID: uuid.NewString(), DeviceID: uuid.NewString(), EnvironmentID: uuid.NewString(), TenantID: uuid.NewString(), ProviderKey: uuid.NewString(), State: "running", Expired: true, CreateSettled: true} + for _, deleted := range []bool{false, true} { + t.Run(fmt.Sprint("deleted=", deleted), func(t *testing.T) { + current := owner + current.SessionDeleted = deleted + checked, cancelled, revoked := false, false, false + tx := &fakeCleanupTx{fakeAllocationTx: &fakeAllocationTx{t: t, loadAllocation: func() (Allocation, error) { return current, nil }}, + canRetainEnvironment: func(Allocation) (bool, error) { + if !revoked { + t.Fatal("eligibility checked before authority revocation") + } + checked = true + return true, nil + }, + revokeDevice: func(Allocation) error { revoked = true; return nil }, + requestCleanup: func(a Allocation) (Allocation, error) { a.State = "cleanup_pending"; return a, nil }, + loadActiveTurn: func() (sessions.Turn, bool, error) { cancelled = true; return sessions.Turn{}, false, nil }, cancelPendingInput: func() error { return nil }, + } + storage := &fakeExecutionStorage{t: t, withAllocationCleanup: func(_ context.Context, _ AllocationKey, apply func(AllocationCleanupTx) error) error { + return apply(tx) + }} + operations, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, testPublicURL), storage, engine.Catalog{}) + if err != nil { + t.Fatal(err) + } + result, err := operations.RequestCleanup(t.Context(), current) + if err != nil || result.State != "cleanup_pending" || !revoked || checked == deleted || cancelled != deleted { + t.Fatal("cleanup lost lifetime separation", result, err, checked, cancelled) + } + }) + } +} + +func (f *fakeReservationTx) LoadGenerationSpecification(uint64) (GenerationSpecification, error) { + unexpected(f.t, "LoadGenerationSpecification") + return GenerationSpecification{}, nil +} + +func TestSetComputeRechecksIdlePressureAndActivity(t *testing.T) { + now := time.Now() + owner := Allocation{ID: uuid.NewString(), DeviceID: uuid.NewString(), TenantID: uuid.NewString(), EnvironmentID: uuid.NewString(), NodeID: uuid.NewString(), ProviderKey: uuid.NewString(), State: "running", ComputePhase: "running", ComputeActivityAt: now.Add(-time.Minute)} + for _, test := range []struct { + name string + idle time.Duration + busy, wake, changed, pressure bool + want error + checkPressure bool + }{ + {name: "idle timeout", idle: 6 * time.Minute}, + {name: "early pressure", idle: time.Minute, pressure: true, checkPressure: true}, + {name: "no demand", idle: time.Minute, checkPressure: true, want: ErrNotIdle}, + {name: "quiet grace", idle: time.Second, pressure: true, want: ErrNotIdle}, + {name: "turn became busy", idle: time.Minute, busy: true, pressure: true, want: ErrAllocationConflict}, + {name: "wake arrived", idle: time.Minute, wake: true, pressure: true, want: ErrAllocationConflict}, + {name: "activity changed", idle: time.Minute, changed: true, pressure: true, want: ErrAllocationConflict}, + } { + t.Run(test.name, func(t *testing.T) { + current := owner + if test.changed { + current.ComputeActivityAt = now + } + pressureChecks, writes := 0, 0 + tx := &fakeAllocationTx{t: t, + loadAllocation: func() (Allocation, error) { return current, nil }, + loadSessionDevice: func() (SessionDevice, bool, error) { + return SessionDevice{ID: owner.DeviceID, EnvironmentID: owner.EnvironmentID}, true, nil + }, + loadActivity: func(Allocation) (Activity, error) { + return Activity{ObservedAt: now, LastActivity: now.Add(-test.idle), Busy: test.busy, WakeRequested: test.wake}, nil + }, + loadSuspensionDemand: func(Allocation) (SuspensionDemand, error) { + pressureChecks++ + node := placement.Node{ID: owner.NodeID, Online: true, Active: 1, MaxActive: 1, Retained: 1, MaxRetained: 1, CoreURL: testPublicURL} + demand := SuspensionDemand{Deployment: placement.Deployment{InstallationID: owner.ProviderKey, Mode: "nodes"}, Nodes: []placement.Node{node}} + if test.pressure { + compatibility := sandbox.CheckpointCompatibility{ArtifactDomain: "fixture", ExecutionClass: "fixture"} + demand.CheckpointRestores = []CheckpointRestoreDemand{{SourceNode: owner.NodeID, Compatibility: compatibility, Nodes: []placement.CheckpointNode{{Node: node, Checkpoint: &compatibility}}}} + } + return demand, nil + }, + setCompute: func(a Allocation, c ComputeChange) (Allocation, error) { + writes++ + a.ComputePhase = c.Phase + return a, nil + }, + } + storage := &fakeExecutionStorage{t: t, withAllocation: func(_ context.Context, _ AllocationKey, apply func(AllocationTx) error) error { return apply(tx) }} + operations, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, testPublicURL), storage, engine.Catalog{}) + if err != nil { + t.Fatal(err) + } + until := now.Add(time.Hour) + _, err = operations.SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute) + if !errors.Is(err, test.want) { + t.Fatalf("SetCompute: %v, want %v", err, test.want) + } + if (pressureChecks == 1) != test.checkPressure || (writes == 1) != (test.want == nil) { + t.Fatalf("pressure checks=%d writes=%d", pressureChecks, writes) + } + }) + } +} + +func (f *fakeAllocationTx) CanRetainEnvironment(current Allocation) (bool, error) { + if f.canRetainEnvironment == nil { + return false, nil + } + return f.canRetainEnvironment(current) +} + +func TestBeginRestoreDirectKeepsAllocationAndRetention(t *testing.T) { + until := time.Now().Add(time.Hour) + owner := Allocation{ID: uuid.NewString(), DeviceID: uuid.NewString(), TenantID: uuid.NewString(), EnvironmentID: uuid.NewString(), ProviderKey: uuid.NewString(), State: "running", ComputePhase: "suspended", ComputeRetainedUntil: &until} + tx := &fakeAllocationTx{t: t, loadAllocation: func() (Allocation, error) { return owner, nil }, loadSessionDevice: func() (SessionDevice, bool, error) { + return SessionDevice{ID: owner.DeviceID, EnvironmentID: owner.EnvironmentID}, true, nil + }, loadActivity: func(Allocation) (Activity, error) { return Activity{WakeRequested: true}, nil }, setCompute: func(current Allocation, change ComputeChange) (Allocation, error) { + if change.Phase != "restoring" || change.RetainedUntil != owner.ComputeRetainedUntil { + t.Fatal("direct restore lost intent", change) + } + current.ComputePhase = change.Phase + return current, nil + }} + storage := &fakeExecutionStorage{t: t, withAllocation: func(_ context.Context, _ AllocationKey, apply func(AllocationTx) error) error { return apply(tx) }} + operations, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, testPublicURL), storage, engine.Catalog{}) + if err != nil { + t.Fatal(err) + } + restored, err := operations.BeginRestore(t.Context(), owner, sandbox.CheckpointCompatibility{}, json.RawMessage(`{"restore_id":"attempt"}`)) + if err != nil || restored.ID != owner.ID || restored.NodeID != "" || restored.ComputePhase != "restoring" { + t.Fatal("direct restore changed ownership", restored, err) + } + if _, err := operations.SetCompute(t.Context(), owner, "restoring", json.RawMessage(`{}`), &until, time.Minute); !errors.Is(err, ErrInvalidInput) { + t.Fatal("restore bypassed capacity admission", err) + } +} diff --git a/services/core/internal/deployment/errors.go b/services/core/internal/deployment/errors.go index e9001c419..d3d30bb50 100644 --- a/services/core/internal/deployment/errors.go +++ b/services/core/internal/deployment/errors.go @@ -27,6 +27,8 @@ var ( // matches the stored allocation, device binding, state or compute revision, // or a replay for another installation. ErrAllocationConflict = errors.New("the sandbox allocation changed") + // ErrNotIdle means neither idle expiry nor current demand requires suspension. + ErrNotIdle = errors.New("sandbox compute is not ready to suspend") // ErrNodeAddressMismatch rejects an enrollment whose Core address is not // the installation public URL. The token stays unconsumed. ErrNodeAddressMismatch = errors.New("sandbox node Core address differs from the public URL") diff --git a/services/core/internal/deployment/execution.go b/services/core/internal/deployment/execution.go index e2541d406..40d420e45 100644 --- a/services/core/internal/deployment/execution.go +++ b/services/core/internal/deployment/execution.go @@ -6,6 +6,7 @@ import ( "errors" "math" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" ) @@ -16,15 +17,16 @@ import ( type ExecutionOperations struct { service *Service storage ExecutionStorage + engines engine.Catalog } // NewExecutionOperations binds the deployment changes to the lease-bound // storage of one execution owner. -func NewExecutionOperations(service *Service, storage ExecutionStorage) (*ExecutionOperations, error) { +func NewExecutionOperations(service *Service, storage ExecutionStorage, engines engine.Catalog) (*ExecutionOperations, error) { if service == nil || storage == nil { return nil, errors.New("deployment execution operations require the deployment service and execution storage") } - return &ExecutionOperations{service: service, storage: storage}, nil + return &ExecutionOperations{service: service, storage: storage, engines: engines}, nil } // Claim reserves the installation for Web setup, once per execution owner diff --git a/services/core/internal/deployment/fakes_test.go b/services/core/internal/deployment/fakes_test.go index 0317764a1..4ab3c354b 100644 --- a/services/core/internal/deployment/fakes_test.go +++ b/services/core/internal/deployment/fakes_test.go @@ -662,3 +662,10 @@ func (f *fakeReader) CountRetainedAllocations(ctx context.Context, installationI } return f.countRetainedAllocations(ctx, installationID) } + +func (f *fakeReader) PlacementDemand(context.Context, PlacementDemandCursor) ([]PlacementDemand, PlacementDemandCursor, error) { + panic("unexpected PlacementDemand") +} +func (f *fakeReader) RetainedNativeHistory(context.Context, AllocationKey) (bool, error) { + panic("unexpected RetainedNativeHistory") +} diff --git a/services/core/internal/deployment/nodes.go b/services/core/internal/deployment/nodes.go index f166a12b8..94c281f5b 100644 --- a/services/core/internal/deployment/nodes.go +++ b/services/core/internal/deployment/nodes.go @@ -11,6 +11,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/coremetrics" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/providercontract" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/google/uuid" ) @@ -516,8 +517,8 @@ func (s *Service) NodeRetention(ctx context.Context, nodeID, connectionID string // Heartbeat records a protocol 1 node's health, which reports readiness of its // enrollment generation only. -func (s *Service) Heartbeat(ctx context.Context, nodeID, connectionID string, epoch uint64, health NodeHealth) error { - return s.heartbeat(ctx, nodeID, connectionID, epoch, health, nil, 1) +func (s *Service) Heartbeat(ctx context.Context, nodeID, connectionID string, epoch uint64, health NodeHealth, statuses []sandbox.GenerationStatus) error { + return s.heartbeat(ctx, nodeID, connectionID, epoch, health, statuses, 1) } // HeartbeatGenerations records a node's health and the preparation state of @@ -558,7 +559,12 @@ func (s *Service) heartbeat(ctx context.Context, nodeID, connectionID string, ep if !current { return ErrNodeCredential } - if protocol == 1 { + if protocol == 1 && len(statuses) > 0 { + if len(statuses) != 1 || statuses[0].Generation != n.DeploymentGeneration || statuses[0].SpecificationDigest != n.SpecificationDigest || (statuses[0].State == "ready") != health.ProviderReady { + return ErrInvalidInput + } + } + if protocol == 1 && len(statuses) == 0 { state := "failed" if health.ProviderReady { state = "ready" @@ -570,8 +576,16 @@ func (s *Service) heartbeat(ctx context.Context, nodeID, connectionID string, ep } func (s *Service) recordGenerations(tx NodeTx, d Record, n StoredNode, statuses []sandbox.GenerationStatus, protocol int) error { + adapter, err := s.registry.Lookup(d.Provider) + if err != nil { + return err + } + checkpointSupported := adapter.Operations()["Initial"].State == providercontract.Supported seen := map[uint64]bool{} for _, status := range statuses { + if status.Checkpoint != nil && (status.State != "ready" || !checkpointSupported || status.Checkpoint.Validate() != nil) || status.State == "ready" && checkpointSupported && status.Checkpoint == nil { + return ErrInvalidInput + } if !validGeneration(status.Generation) || !validDigest(status.SpecificationDigest) || seen[status.Generation] || (status.State != "ready" && status.State != "preparing" && status.State != "failed") || status.State == "ready" && status.Diagnostic != "" { return ErrInvalidInput } @@ -592,7 +606,7 @@ func (s *Service) recordGenerations(tx NodeTx, d Record, n StoredNode, statuses if spec.Digest(d.Provider) != status.SpecificationDigest { return ErrSpecificationMismatch } - if err := tx.UpsertGenerationStatus(GenerationStatusRecord{NodeID: n.ID, ConnectionID: n.ConnectionID, Generation: status.Generation, SpecificationDigest: status.SpecificationDigest, OwnerEpoch: d.OwnerEpoch, State: status.State, Diagnostic: sandbox.NormalizeNodeDiagnostic(status.Diagnostic)}); err != nil { + if err := tx.UpsertGenerationStatus(GenerationStatusRecord{NodeID: n.ID, ConnectionID: n.ConnectionID, Generation: status.Generation, SpecificationDigest: status.SpecificationDigest, OwnerEpoch: d.OwnerEpoch, State: status.State, Diagnostic: sandbox.NormalizeNodeDiagnostic(status.Diagnostic), Checkpoint: status.Checkpoint}); err != nil { return err } if status.State == "ready" && status.Generation == d.Generation { diff --git a/services/core/internal/deployment/placement/placement.go b/services/core/internal/deployment/placement/placement.go index d68b8e153..151deea6a 100644 --- a/services/core/internal/deployment/placement/placement.go +++ b/services/core/internal/deployment/placement/placement.go @@ -109,14 +109,41 @@ type Reserved struct { Available bool } -// Restore is what restoring suspended compute on its node reads: the node, -// nil when the installation no longer has it, and whether the node is ready -// for the allocation's generation. -type Restore struct { - Node *Node - // Generation is zero when the allocation has none. - Generation uint64 - GenerationReady bool +// CheckpointNode is a fresh readiness declaration for one immutable generation. +// Capacity is read under the same deployment lock as the allocation transfer. +type CheckpointNode struct { + Node Node + Checkpoint *sandbox.CheckpointCompatibility +} + +// ChooseCheckpoint selects a compatible node without interpreting adapter tokens. +// A retained allocation already owns its source slot. Cleanup needs no active +// slot and compares only the artifact domain, but still reserves a retained slot. +func (r *Rules) ChooseCheckpoint(nodes []CheckpointNode, source string, compatibility sandbox.CheckpointCompatibility, cleanup bool) (string, error) { + var chosen *Node + for i := range nodes { + candidate := &nodes[i] + n := &candidate.Node + if !n.Online || candidate.Checkpoint == nil || candidate.Checkpoint.ArtifactDomain != compatibility.ArtifactDomain { + continue + } + if !cleanup && (candidate.Checkpoint.ExecutionClass != compatibility.ExecutionClass || n.CoreURL != r.publicURL || n.Active >= int64(n.MaxActive)) { + continue + } + if n.ID != source && n.Retained >= int64(n.MaxRetained) { + continue + } + if n.ID == source { + return n.ID, nil + } + if chosen == nil || n.Active < chosen.Active || n.Active == chosen.Active && n.Retained < chosen.Retained { + chosen = n + } + } + if chosen == nil { + return "", ErrNodeUnavailable + } + return chosen.ID, nil } // CheckPublicOrigin rejects a provider that requires a reachable public @@ -205,17 +232,6 @@ func CheckReserved(reserved Reserved) error { return nil } -// CheckRestore admits restoring suspended compute on its original node: the -// node must be online, ready for the allocation's generation and below its -// active capacity. Restore never selects another node. -func CheckRestore(restore Restore) error { - n := restore.Node - if n == nil || restore.Generation == 0 || !n.Online || !restore.GenerationReady || n.Active >= int64(n.MaxActive) { - return ErrNodeUnavailable - } - return nil -} - // LoopbackOrigin reports whether a validated origin names a loopback host, // which nothing outside the Core host can reach. func LoopbackOrigin(value string) bool { diff --git a/services/core/internal/deployment/placement/placement_test.go b/services/core/internal/deployment/placement/placement_test.go index 473c343bb..ae9c7442f 100644 --- a/services/core/internal/deployment/placement/placement_test.go +++ b/services/core/internal/deployment/placement/placement_test.go @@ -158,7 +158,7 @@ func TestDecidePlacement(t *testing.T) { } } -func TestCheckReservedAndRestore(t *testing.T) { +func TestCheckReserved(t *testing.T) { for name, test := range map[string]struct { reserved Reserved want error @@ -171,24 +171,6 @@ func TestCheckReservedAndRestore(t *testing.T) { t.Errorf("%s: CheckReserved = %v, want %v", name, err, test.want) } } - node := func(online bool, active int64) *Node { - return &Node{ID: "a", Online: online, Active: active, MaxActive: 2} - } - for name, test := range map[string]struct { - restore Restore - want error - }{ - "ready": {Restore{Node: node(true, 1), Generation: 3, GenerationReady: true}, nil}, - "missing node": {Restore{Generation: 3, GenerationReady: true}, ErrNodeUnavailable}, - "no generation": {Restore{Node: node(true, 1), GenerationReady: true}, ErrNodeUnavailable}, - "offline": {Restore{Node: node(false, 1), Generation: 3, GenerationReady: true}, ErrNodeUnavailable}, - "generation unready": {Restore{Node: node(true, 1), Generation: 3}, ErrNodeUnavailable}, - "at capacity": {Restore{Node: node(true, 2), Generation: 3, GenerationReady: true}, ErrNodeUnavailable}, - } { - if err := CheckRestore(test.restore); !errors.Is(err, test.want) || (test.want == nil) != (err == nil) { - t.Errorf("%s: CheckRestore = %v, want %v", name, err, test.want) - } - } } func TestLoopbackOrigin(t *testing.T) { diff --git a/services/core/internal/deployment/replacement.go b/services/core/internal/deployment/replacement.go new file mode 100644 index 000000000..980ba5bd2 --- /dev/null +++ b/services/core/internal/deployment/replacement.go @@ -0,0 +1,140 @@ +package deployment + +import ( + "context" + "encoding/json" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" +) + +// EnsurePlacement reserves first or retained demand before entering a node lifecycle. +// The immutable engine profile and actual selected generation are validated +// before committing its reservation. +func (e *ExecutionOperations) EnsurePlacement(ctx context.Context, key AllocationKey, installation string) (placement.Reserved, error) { + key, err := key.parse() + if err != nil { + return placement.Reserved{}, err + } + installation, err = parseID(installation) + if err != nil { + return placement.Reserved{}, err + } + var result placement.Reserved + err = e.storage.WithReservation(ctx, key, func(locked sessions.LockedSession, tx ReservationTx) error { + if err := locked.Public(); err != nil { + return err + } + existing, found, err := tx.FindAllocation() + if err != nil { + return err + } + environment, err := tx.LoadEnvironment(ctx) + if err != nil { + return err + } + kind, err := sessions.EnvironmentType(environment.Configuration) + if err != nil || kind != "openai_hosted" || environment.Status == "failed" || environment.Status == "expired" { + return ErrAllocationConflict + } + if found { + if existing.State != "released" { + return ErrAllocationConflict + } + allowed, err := tx.CanReplaceAllocation() + if err != nil { + return err + } + if !allowed { + return ErrAllocationConflict + } + } else if environment.Initialization != "pending" && environment.Initialization != "complete" { + return ErrAllocationConflict + } + d, err := tx.LockDeployment() + if err != nil { + return err + } + if err := e.service.rules.CheckAdmission(d, installation); err != nil { + return err + } + if d.Mode != string(sandbox.DeploymentNodes) { + return ErrAllocationConflict + } + reserved, err := tx.LoadReserved() + if err != nil { + return err + } + if !reserved.Released { + if err := placement.CheckReserved(reserved); err != nil { + return err + } + result = reserved + return nil + } + nodes, err := tx.LoadNodes() + if err != nil { + return err + } + filtered, err := e.eligibleNodes(d, nodes, locked.Engine, found, tx.LoadGenerationSpecification) + if err != nil { + return err + } + + selected, err := e.service.rules.DecidePlacement(d, filtered) + if err != nil { + return err + } + if selected == nil { + return ErrAllocationConflict + } + if err := tx.ReservePlacement(*selected); err != nil { + return err + } + result = placement.Reserved{NodeID: selected.NodeID, Generation: selected.Generation, Available: true} + return nil + }) + return result, err +} + +// eligibleNodes applies one generation and Harness compatibility decision for +// allocation admission and capacity-pressure decisions. Retained files require +// another external-storage generation; new Sessions may select either kind. +func (e *ExecutionOperations) eligibleNodes(d placement.Deployment, nodes []placement.Node, engineKind string, retained bool, loadGeneration func(uint64) (GenerationSpecification, error)) ([]placement.Node, error) { + profile, known := e.engines.Lookup(engineKind) + if !known { + return nil, sessions.ErrInvalidInput + } + compatible := make(map[uint64]bool) + filtered := make([]placement.Node, 0, len(nodes)) + for _, node := range nodes { + if node.ReadyGeneration == nil { + // Placement cannot select it, but preserves its preparing diagnosis. + filtered = append(filtered, node) + continue + } + generation := *node.ReadyGeneration + accepts, known := compatible[generation] + if !known { + specification := d.Specification + if generation != d.Generation { + stored, err := loadGeneration(generation) + if err != nil { + return nil, err + } + specification = stored.Specification + } + var spec sandbox.DeploymentSpec + if err := json.Unmarshal(specification, &spec); err != nil { + return nil, err + } + accepts = (!retained || spec.Workspace != nil) && sessions.ValidateRetainedHistory(spec.Workspace != nil, profile.RetainedNativeHistory.IsSupported()) == nil + compatible[generation] = accepts + } + if accepts { + filtered = append(filtered, node) + } + } + return filtered, nil +} diff --git a/services/core/internal/deployment/service_test.go b/services/core/internal/deployment/service_test.go index e3554720f..ae703bb6a 100644 --- a/services/core/internal/deployment/service_test.go +++ b/services/core/internal/deployment/service_test.go @@ -4,6 +4,7 @@ import ( "context" "encoding/json" "errors" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "strings" "testing" "time" @@ -46,7 +47,7 @@ func operations(t *testing.T, publicURL string, tx *fakeDeploymentTx) *Execution if tx != nil { storage.withDeployment = func(_ context.Context, apply func(DeploymentTx) error) error { return apply(tx) } } - result, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, publicURL), storage) + result, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, publicURL), storage, engine.Catalog{}) if err != nil { t.Fatal(err) } @@ -89,10 +90,10 @@ func TestNewServiceAndOperationsRejectNilDependencies(t *testing.T) { } } service := newService(t, storage, reader, testPublicURL) - if result, err := NewExecutionOperations(nil, &fakeExecutionStorage{t: t}); result != nil || err == nil { + if result, err := NewExecutionOperations(nil, &fakeExecutionStorage{t: t}, engine.Catalog{}); result != nil || err == nil { t.Errorf("NewExecutionOperations without the service = %v, %v", result, err) } - if result, err := NewExecutionOperations(service, nil); result != nil || err == nil { + if result, err := NewExecutionOperations(service, nil, engine.Catalog{}); result != nil || err == nil { t.Errorf("NewExecutionOperations without storage = %v, %v", result, err) } } diff --git a/services/core/internal/deployment/session_archive.go b/services/core/internal/deployment/session_archive.go index 0f6c98ad4..22f5d41d6 100644 --- a/services/core/internal/deployment/session_archive.go +++ b/services/core/internal/deployment/session_archive.go @@ -101,7 +101,7 @@ func (e *ExecutionOperations) archiveSession(ctx context.Context, tenantID, sess if err := tx.RequestArchiveCleanup(current); err != nil { return err } - } else if !allocated { + } else { if err := tx.ReleasePlacement(); err != nil { return err } diff --git a/services/core/internal/deployment/session_archive_test.go b/services/core/internal/deployment/session_archive_test.go index d325e6b83..781905129 100644 --- a/services/core/internal/deployment/session_archive_test.go +++ b/services/core/internal/deployment/session_archive_test.go @@ -4,6 +4,7 @@ import ( "context" "encoding/json" "errors" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "slices" "strings" "testing" @@ -216,7 +217,7 @@ func archiveOperations(t *testing.T, tx *fakeArchiveTx, locked sessions.LockedSe } return apply(ctx, locked, tx) }} - operations, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, testPublicURL), storage) + operations, err := NewExecutionOperations(newService(t, &fakeStorage{t: t}, &fakeReader{t: t}, testPublicURL), storage, engine.Catalog{}) if err != nil { t.Fatal(err) } @@ -250,7 +251,7 @@ func TestArchiveSession(t *testing.T) { join(head, expire, []string{"RequestArchiveCleanup device allocation", "RecordArchiveAudit", "LoadArchive"})}, {"no allocation", sessions.LockedSession{}, hosted, nil, nil, join(head, expire, []string{"ReleasePlacement", "RecordArchiveAudit", "LoadArchive"})}, {"released allocation", sessions.LockedSession{}, hosted, &Allocation{ID: "allocation", ProviderKey: "previous", State: "released"}, nil, - join(head, expire, []string{"RecordArchiveAudit", "LoadArchive"})}, + join(head, expire, []string{"ReleasePlacement", "RecordArchiveAudit", "LoadArchive"})}, {"ended Environment", sessions.LockedSession{}, expired, live, nil, join(head, settle, []string{"RequestArchiveCleanup device allocation", "RecordArchiveAudit", "LoadArchive"})}, {"allocation of another installation", sessions.LockedSession{}, hosted, &Allocation{ID: "allocation", ProviderKey: "previous", State: "running"}, ErrConflict, head}, diff --git a/services/core/internal/deployment/storage.go b/services/core/internal/deployment/storage.go index 2364910bb..c15b778ca 100644 --- a/services/core/internal/deployment/storage.go +++ b/services/core/internal/deployment/storage.go @@ -96,12 +96,24 @@ type ReservationTx interface { LoadReserved() (placement.Reserved, error) // InsertAllocation stores the allocation of the Session's Environment. InsertAllocation(allocation NewAllocation) (Allocation, error) + // CanReplaceAllocation verifies retained storage and settled, revoked prior ownership. + CanReplaceAllocation() (bool, error) + // LoadGenerationSpecification reads immutable storage requirements for placement. + LoadGenerationSpecification(generation uint64) (GenerationSpecification, error) + // LoadNodes reads eligible placement facts under the deployment lock. + LoadNodes() ([]placement.Node, error) + // ReservePlacement inserts a first reservation or replaces a released reservation without changing historical receipts. + ReservePlacement(placement.Placement) error } // AllocationTx is one Session-locked allocation change. Each write applies to // current, the allocation LoadAllocation returned, and returns it changed. A // write whose stored guard no longer holds is ErrAllocationConflict. type AllocationTx interface { + // CanRetainEnvironment checks the ready retained filesystem, completed + // initialization and selected Harness's qualified history under the Session lock. + CanRetainEnvironment(current Allocation) (bool, error) + // LoadAllocation returns the Environment's allocation. LoadAllocation() (Allocation, error) // LoadSessionDevice returns the device the Session is bound to and @@ -109,9 +121,19 @@ type AllocationTx interface { LoadSessionDevice() (SessionDevice, bool, error) // LoadActivity returns the allocation's activity. LoadActivity(current Allocation) (Activity, error) - // LoadRestore locks the deployment and returns what restoring the - // allocation's suspended compute on its node reads. - LoadRestore(current Allocation) (placement.Restore, error) + // LoadSuspensionDemand locks the deployment and reads demand and capacity. + LoadSuspensionDemand(current Allocation) (SuspensionDemand, error) + // PlacementDemand reads one qualified page; next advances past all inspected + // rows, including ineligible receipts. An empty next.EnvironmentID ends the scan. + PlacementDemand(after PlacementDemandCursor) ([]PlacementDemand, PlacementDemandCursor, error) + // LoadGenerationSpecification reads immutable generation requirements. + LoadGenerationSpecification(generation uint64) (GenerationSpecification, error) + // LoadCheckpointPlacement locks the deployment and loads fresh readiness + // for the allocation's exact immutable generation, including capacity. + LoadCheckpointPlacement(current Allocation) (placement.Deployment, []placement.CheckpointNode, error) + // MoveSuspended changes node ownership, placement and compute intent in one + // transaction. Only a settled suspended allocation can change nodes. + MoveSuspended(current Allocation, nodeID string, change ComputeChange) (Allocation, error) // ObserveRunning records the allocation running with its creation settled. ObserveRunning(current Allocation) (Allocation, error) // SettleCreation records that the original Create can no longer change @@ -210,7 +232,7 @@ type Reader interface { // AddressBindings counts what is bound to an installation address, read // in one snapshot. publicURL is the address nodes are compared against. AddressBindings(ctx context.Context, publicURL string) (AddressBindings, error) - // EnvironmentAllocation returns the Environment's allocation, including + // EnvironmentAllocation returns the Environment's latest allocation receipt, including // for a deleted Session. It returns ErrInvalidInput for a malformed // identifier and ErrNotFound for a missing allocation. EnvironmentAllocation(ctx context.Context, key AllocationKey) (Allocation, error) @@ -240,6 +262,10 @@ type Reader interface { // ID, in ID order. It reads the committed placement and never selects a // replacement node. UnallocatedEnvironments(ctx context.Context, nodeID, after string) ([]UnallocatedEnvironment, error) + // RetainedNativeHistory reports the latest owner's persisted retention eligibility. + RetainedNativeHistory(ctx context.Context, key AllocationKey) (bool, error) + // PlacementDemand lists up to 32 unreserved first or retained requests in demand-time and ID order. + PlacementDemand(ctx context.Context, after PlacementDemandCursor) ([]PlacementDemand, PlacementDemandCursor, error) // LifecyclePlacement returns what routing the Environment to its // lifecycle reads, including for a deleted Session. A missing Environment // is sessions.ErrNotFound. @@ -491,4 +517,5 @@ type GenerationStatusRecord struct { SpecificationDigest string OwnerEpoch uint64 State, Diagnostic string + Checkpoint *sandbox.CheckpointCompatibility } diff --git a/services/core/internal/deployment/suspension_pressure.go b/services/core/internal/deployment/suspension_pressure.go new file mode 100644 index 000000000..d8ae4c173 --- /dev/null +++ b/services/core/internal/deployment/suspension_pressure.go @@ -0,0 +1,148 @@ +package deployment + +import ( + "errors" + "slices" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +// SuspensionDemand is a transaction-bound capacity observation. It is not a +// reservation: only a confirmed suspension frees an active slot. +type SuspensionDemand struct { + Deployment placement.Deployment + Nodes []placement.Node + InFlightNodes []string + CheckpointRestores []CheckpointRestoreDemand +} + +// CheckpointRestoreDemand carries one stopped owner and exact-generation target evidence. +type CheckpointRestoreDemand struct { + SourceNode string + Compatibility sandbox.CheckpointCompatibility + Nodes []placement.CheckpointNode +} + +func (e *ExecutionOperations) canSuspendForDemand(tx AllocationTx, owner Allocation) (bool, error) { + pressure, err := tx.LoadSuspensionDemand(owner) + if err != nil { + return false, err + } + if pressure.Deployment.Resetting || pressure.Deployment.Mode != string(sandbox.DeploymentNodes) || pressure.Deployment.InstallationID != owner.ProviderKey { + return false, nil + } + var source *placement.Node + for i := range pressure.Nodes { + if pressure.Nodes[i].ID == owner.NodeID { + source = &pressure.Nodes[i] + break + } + } + if source == nil || !source.Online || source.Active < int64(source.MaxActive) { + return false, nil + } + for _, demand := range pressure.CheckpointRestores { + if _, err := e.service.rules.ChooseCheckpoint(demand.Nodes, demand.SourceNode, demand.Compatibility, false); err == nil { + continue + } else if !errors.Is(err, placement.ErrNodeUnavailable) { + return false, err + } + candidates := slices.Clone(demand.Nodes) + for i := range candidates { + if candidates[i].Node.ID == owner.NodeID { + candidates[i].Node.Active-- + } + } + chosen, err := e.service.rules.ChooseCheckpoint(candidates, demand.SourceNode, demand.Compatibility, false) + if err == nil && chosen == owner.NodeID { + eligible := make([]placement.Node, 0, len(candidates)) + for _, candidate := range candidates { + if candidate.Checkpoint != nil && *candidate.Checkpoint == demand.Compatibility { + eligible = append(eligible, candidate.Node) + } + } + inFlight, err := e.reclamationInFlight(pressure, eligible) + if err != nil { + return false, err + } + if !inFlight { + return true, nil + } + } else if err != nil && !errors.Is(err, placement.ErrNodeUnavailable) { + return false, err + } + } + if source.Retained >= int64(source.MaxRetained) { + return false, nil + } + return e.placementNeedsCapacity(tx, pressure, owner.NodeID) +} + +// placementNeedsCapacity scans bounded pages under the caller's deployment +// lock and transaction deadline. Suspension must make active capacity usable +// without releasing a retained slot; an ineligible receipt supplies no pressure. +func (e *ExecutionOperations) placementNeedsCapacity(tx AllocationTx, pressure SuspensionDemand, nodeID string) (bool, error) { + var after PlacementDemandCursor + for { + demands, next, err := tx.PlacementDemand(after) + if err != nil { + return false, err + } + for _, demand := range demands { + nodes, err := e.eligibleNodes(pressure.Deployment, pressure.Nodes, demand.Engine, demand.Retained, tx.LoadGenerationSpecification) + if err != nil { + return false, err + } + if _, err := e.service.rules.DecidePlacement(pressure.Deployment, nodes); err == nil { + // Let the demand scan consume real free capacity first. + return false, nil + } else if !errors.Is(err, placement.ErrNodeUnavailable) && !errors.Is(err, placement.ErrNodesPreparing) { + return false, err + } + inFlight, err := e.reclamationInFlight(pressure, nodes) + if err != nil { + return false, err + } + if inFlight { + continue + } + for i := range nodes { + if nodes[i].ID == nodeID { + nodes[i].Active-- + } + } + chosen, err := e.service.rules.DecidePlacement(pressure.Deployment, nodes) + if err == nil && chosen != nil && chosen.NodeID == nodeID { + return true, nil + } + if err != nil && !errors.Is(err, placement.ErrNodeUnavailable) && !errors.Is(err, placement.ErrNodesPreparing) { + return false, err + } + } + if next.EnvironmentID == "" { + return false, nil + } + after = next + } +} + +// reclamationInFlight coordinates only resources whose nodes can serve this +// demand. Unavailable or incompatible nodes retain their receipts without +// stopping pressure recovery on a healthy node. +func (e *ExecutionOperations) reclamationInFlight(pressure SuspensionDemand, nodes []placement.Node) (bool, error) { + for _, node := range nodes { + if !slices.Contains(pressure.InFlightNodes, node.ID) { + continue + } + // Capacity is what the in-flight operation may release; all other + // placement requirements must hold now, using the normal admission rule. + node.Active, node.Retained = 0, 0 + if chosen, err := e.service.rules.DecidePlacement(pressure.Deployment, []placement.Node{node}); err == nil && chosen != nil { + return true, nil + } else if err != nil && !errors.Is(err, placement.ErrNodeUnavailable) && !errors.Is(err, placement.ErrNodesPreparing) { + return false, err + } + } + return false, nil +} diff --git a/services/core/internal/engine/claude.go b/services/core/internal/engine/claude.go index 7a19d266c..b92c435af 100644 --- a/services/core/internal/engine/claude.go +++ b/services/core/internal/engine/claude.go @@ -13,6 +13,7 @@ import ( // providers reject text without non-whitespace characters. func claudeProfile() Profile { return Profile{ + RetainedNativeHistory: proto.CapabilitySupported, ProgrammaticToolCallingDisable: proto.CapabilitySupported, MCPOrigins: []string{"service", "environment"}, Placements: []string{"none", "openai_hosted", "self_hosted"}, diff --git a/services/core/internal/engine/codex.go b/services/core/internal/engine/codex.go index a2f67bc07..7ce42df97 100644 --- a/services/core/internal/engine/codex.go +++ b/services/core/internal/engine/codex.go @@ -7,6 +7,7 @@ import ( func codexProfile() Profile { return Profile{ + RetainedNativeHistory: proto.CapabilitySupported, ProgrammaticToolCallingDisable: proto.CapabilitySupported, Placements: []string{"none", "self_hosted", "openai_hosted"}, MCPOrigins: []string{"service", "environment"}, diff --git a/services/core/internal/engine/enginetest/profile.go b/services/core/internal/engine/enginetest/profile.go index 2984c5e93..1a5801474 100644 --- a/services/core/internal/engine/enginetest/profile.go +++ b/services/core/internal/engine/enginetest/profile.go @@ -11,6 +11,7 @@ import ( // intentionally exhaustive; new fields stay unspecified until decided here. func Profile(change func(*engine.Profile)) engine.Profile { p := engine.Profile{ + RetainedNativeHistory: proto.CapabilityUnsupported, ProgrammaticToolCallingDisable: proto.CapabilityUnsupported, WebSearchControl: proto.CapabilityUnsupported, TextVerbosity: proto.CapabilityUnsupported, diff --git a/services/core/internal/engine/mcode.go b/services/core/internal/engine/mcode.go index 950cc470e..eea6c9cf0 100644 --- a/services/core/internal/engine/mcode.go +++ b/services/core/internal/engine/mcode.go @@ -12,6 +12,7 @@ import ( // with "Local message content or attachments are required.". func mcodeProfile() Profile { return Profile{ + RetainedNativeHistory: proto.CapabilityUnsupported, MCPOrigins: []string{"environment"}, Placements: []string{"none", "openai_hosted", "self_hosted"}, MCPBearer: proto.CapabilitySupported, diff --git a/services/core/internal/engine/profile.go b/services/core/internal/engine/profile.go index afb701cc7..501ef240c 100644 --- a/services/core/internal/engine/profile.go +++ b/services/core/internal/engine/profile.go @@ -16,6 +16,7 @@ var ErrInvalidInput = errors.New("invalid engine configuration") // Profile records qualified public behavior, independently of Runtime advertisements. type Profile struct { + RetainedNativeHistory proto.CapabilitySupport ProgrammaticToolCallingDisable proto.CapabilitySupport Placements []string MCPOrigins []string diff --git a/services/core/internal/execution/deployment_fixture_test.go b/services/core/internal/execution/deployment_fixture_test.go index 67716db1b..0aa389d13 100644 --- a/services/core/internal/execution/deployment_fixture_test.go +++ b/services/core/internal/execution/deployment_fixture_test.go @@ -3,6 +3,7 @@ package execution import ( "context" "errors" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "testing" "time" @@ -68,7 +69,7 @@ func deploymentOperations(t *testing.T, storage deployment.Storage, reader deplo if err != nil { t.Fatal(err) } - operations, err := deployment.NewExecutionOperations(service, execution) + operations, err := deployment.NewExecutionOperations(service, execution, engine.Catalog{}) if err != nil { t.Fatal(err) } @@ -142,6 +143,8 @@ func (s *strictExecutionStorage) WithDeployment(ctx context.Context, apply func( type strictDeploymentReader struct { countRetainedAllocations func(context.Context, string) (int64, error) countComputeReservations func(context.Context, string) (int64, error) + placementDemand func(context.Context, deployment.PlacementDemandCursor) ([]deployment.PlacementDemand, deployment.PlacementDemandCursor, error) + retainedNativeHistory func(context.Context, deployment.AllocationKey) (bool, error) t *testing.T deployment func(context.Context) (deployment.Record, error) snapshot func(context.Context) (deployment.Snapshot, error) @@ -357,3 +360,16 @@ func (r *strictDeploymentReader) CountRetainedAllocations(ctx context.Context, i } return r.countRetainedAllocations(ctx, installationID) } + +func (r *strictDeploymentReader) PlacementDemand(ctx context.Context, after deployment.PlacementDemandCursor) ([]deployment.PlacementDemand, deployment.PlacementDemandCursor, error) { + if r.placementDemand == nil { + return nil, deployment.PlacementDemandCursor{}, unexpectedDeploymentCall(r.t, "PlacementDemand") + } + return r.placementDemand(ctx, after) +} +func (r *strictDeploymentReader) RetainedNativeHistory(ctx context.Context, key deployment.AllocationKey) (bool, error) { + if r.retainedNativeHistory == nil { + return false, unexpectedDeploymentCall(r.t, "RetainedNativeHistory") + } + return r.retainedNativeHistory(ctx, key) +} diff --git a/services/core/internal/execution/engine_profile_test.go b/services/core/internal/execution/engine_profile_test.go index 754e370c9..8a41be47b 100644 --- a/services/core/internal/execution/engine_profile_test.go +++ b/services/core/internal/execution/engine_profile_test.go @@ -149,3 +149,31 @@ func TestCommonOnlyValidationPreservesFunctionResults(t *testing.T) { t.Fatal("common-only result acquired a native restriction", err) } } + +func TestRetainedNativeHistoryDependsOnExternalWorkspace(t *testing.T) { + for _, required := range []bool{false, true} { + for _, supported := range []bool{false, true} { + if err := sessions.ValidateRetainedHistory(required, supported); (err == nil) != (!required || supported) { + t.Fatal(required, supported, err) + } + } + } + for _, qualified := range []bool{false, true} { + profile := enginetest.Profile(func(p *engine.Profile) { + p.Placements = []string{"openai_hosted"} + p.RetainedNativeHistory = proto.CapabilityFromBool(qualified) + }) + policy := Policy{Engines: engine.NewCatalog(map[string]engine.Profile{"test_harness": profile})} + if err := policy.ValidateSessionConfiguration("test_harness", json.RawMessage(`{"agent":{"model":"fixture"},"environment":{"type":"openai_hosted"}}`)); err != nil { + t.Fatal("generic hosted configuration acquired a storage requirement", qualified, err) + } + for _, external := range []bool{false, true} { + for _, runtimeSupported := range []bool{false, true} { + err := policy.validateEnvironmentHistory("test_harness", sessions.Environment{ExternalWorkspace: external}, runtimeSupported) + if (err == nil) != (!external || qualified && runtimeSupported) { + t.Fatal(external, qualified, runtimeSupported, err) + } + } + } + } +} diff --git a/services/core/internal/execution/prepared_dispatch.go b/services/core/internal/execution/prepared_dispatch.go index a3b075694..69d901253 100644 --- a/services/core/internal/execution/prepared_dispatch.go +++ b/services/core/internal/execution/prepared_dispatch.go @@ -58,7 +58,7 @@ func (d *Dispatcher) RunEnvironmentInput(ctx context.Context, lease Ownership, t if err != nil { return run, err } - caps, err := d.engineCapabilities(peer, session.Engine, snapshot) + caps, err := d.sessionCapabilities(ctx, peer, session, snapshot) if err != nil { return run, err } diff --git a/services/core/internal/execution/runtime_capabilities_test.go b/services/core/internal/execution/runtime_capabilities_test.go index c2c29bca1..63319df84 100644 --- a/services/core/internal/execution/runtime_capabilities_test.go +++ b/services/core/internal/execution/runtime_capabilities_test.go @@ -40,7 +40,7 @@ func TestRuntimeCapabilitiesPreserveRawBundlesAndSetupOrdering(t *testing.T) { owner := agentcapabilities.Identity{EnvironmentID: uuid.NewString(), SessionID: uuid.NewString()} for _, op := range operations { peer := &capabilityFixture{outcome: "completed"} - if err := runRuntimeSetup(t.Context(), peer, owner, op); err != nil { + if err := runRuntimeSetup(initializationTestContext(t), peer, owner, op); err != nil { t.Fatal(err) } if _, err := uuid.Parse(peer.requestID); err != nil { @@ -64,14 +64,14 @@ func TestRuntimeCapabilitiesPreserveRawBundlesAndSetupOrdering(t *testing.T) { func TestRuntimeCapabilitiesConfirmedAndUnknownFailures(t *testing.T) { for _, outcome := range []string{"failed", "rejected", "unknown", "unexpected"} { peer := &capabilityFixture{outcome: outcome} - err := runRuntimeSetup(t.Context(), peer, agentcapabilities.Identity{}, runtimeSetupOperation{Request: proto.RuntimePreparePayload{Action: "finalize"}}) + err := runRuntimeSetup(initializationTestContext(t), peer, agentcapabilities.Identity{}, runtimeSetupOperation{Request: proto.RuntimePreparePayload{Action: "finalize"}}) var confirmed *runtimeStepFailure if err == nil || errors.As(err, &confirmed) != (outcome == "failed" || outcome == "rejected") { t.Fatal(outcome, err) } } peer := &capabilityFixture{outcome: "completed", err: errors.New(setupCanary)} - err := runRuntimeSetup(t.Context(), peer, agentcapabilities.Identity{}, runtimeSetupOperation{Request: proto.RuntimePreparePayload{}}) + err := runRuntimeSetup(initializationTestContext(t), peer, agentcapabilities.Identity{}, runtimeSetupOperation{Request: proto.RuntimePreparePayload{}}) if err == nil || bytes.Contains([]byte(err.Error()), []byte(setupCanary)) { t.Fatal("transport error leaked or succeeded", err) } diff --git a/services/core/internal/execution/runtime_compute.go b/services/core/internal/execution/runtime_compute.go index 64731c071..e95304682 100644 --- a/services/core/internal/execution/runtime_compute.go +++ b/services/core/internal/execution/runtime_compute.go @@ -51,7 +51,8 @@ func (r *runtimeLifecycle) saveCompute(ctx context.Context, owner deployment.All } return r.deployment.SetCompute(ctx, owner, phase, raw, until, idleTimeout) } -func (r *runtimeLifecycle) enableCompute(ctx context.Context, owner deployment.Allocation) error { +func (r *runtimeLifecycle) enableCompute(ctx context.Context, owner deployment.Allocation) (err error) { + defer func() { err = withObservationOwner(owner, err) }() p := r.config.Provider if err := providercontract.Require(p, "Initial"); err != nil { return err @@ -71,7 +72,8 @@ func (r *runtimeLifecycle) enableCompute(ctx context.Context, owner deployment.A return err } -func (r *runtimeLifecycle) observeCompute(ctx context.Context, owner deployment.Allocation) error { +func (r *runtimeLifecycle) observeCompute(ctx context.Context, owner deployment.Allocation) (err error) { + defer func() { err = withObservationOwner(owner, err) }() p := r.config.Provider if err := providercontract.Require(p, "Initial"); err != nil { return err @@ -111,7 +113,8 @@ func (r *runtimeLifecycle) observeCompute(ctx context.Context, owner deployment. } } -func (r *runtimeLifecycle) idleCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute) error { +func (r *runtimeLifecycle) idleCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute) (err error) { + defer func() { err = withObservationOwner(owner, err) }() compute, err := p.GetCompute(ctx, runtimeReference(owner), state.Current) if err != nil { return err @@ -142,15 +145,19 @@ func (r *runtimeLifecycle) idleCompute(ctx context.Context, p sandbox.SandboxPro return r.deployment.ClearWake(ctx, owner, owner.ComputeActivityAt) } policy := r.config.Suspension - if policy == nil || !activity.ReadyToSuspend(policy.IdleTimeout) { + if policy == nil || activity.Busy { return nil } state.SuspendID, state.RestoreID, state.Rollback = uuid.NewString(), "", false until := activity.ObservedAt.Add(policy.Retention) next, err := r.saveCompute(ctx, owner, "quiescing", state, &until) + if errors.Is(err, deployment.ErrNotIdle) { + return nil + } if err != nil { return err } + owner = next result, err := peer.SuspendControl(ctx, proto.TypeEnvironmentQuiesce, proto.EnvironmentSuspendPayload{EnvironmentID: owner.EnvironmentID, SuspendID: state.SuspendID}) if err != nil { return err @@ -185,13 +192,15 @@ func (r *runtimeLifecycle) idleCompute(ctx context.Context, p sandbox.SandboxPro return r.captureCompute(ctx, p, suspending, state, false) } -func (r *runtimeLifecycle) captureCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute, observeOnly bool) error { +func (r *runtimeLifecycle) captureCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute, observeOnly bool) (err error) { + defer func() { err = withObservationOwner(owner, err) }() result, err := p.Suspend(ctx, sandbox.SuspendRequest{Reference: runtimeReference(owner), OperationID: state.SuspendID, Source: state.Current, Retained: state.Retained, ReconcileOnly: observeOnly}) if err != nil { return err } - if err := sandbox.ValidateSuspendResult(sandbox.SuspendRequest{Reference: runtimeReference(owner), OperationID: state.SuspendID, Source: state.Current, Retained: state.Retained, ReconcileOnly: observeOnly}, result); err != nil { - return err + validationErr := sandbox.ValidateSuspendResult(sandbox.SuspendRequest{Reference: runtimeReference(owner), OperationID: state.SuspendID, Source: state.Current, Retained: state.Retained, ReconcileOnly: observeOnly}, result) + if validationErr != nil && (result.Retained == nil || !errors.Is(validationErr, sandbox.ErrComputeUnconfirmed)) { + return validationErr } if result.Retained == nil { if !observeOnly || !result.SuspendSettled || result.ResourcesReleased || (result.Status != "running" && result.Status != "paused") { @@ -204,12 +213,27 @@ func (r *runtimeLifecycle) captureCompute(ctx context.Context, p sandbox.Sandbox } return r.wakeCompute(ctx, p, next, state) } + if owner.NodeID != "" && result.Retained.Compatibility.Validate() != nil { + return sandbox.ErrOwnership + } state.Retained = result.Retained - _, err = r.saveCompute(ctx, owner, "suspended", state, owner.ComputeRetainedUntil) + // Store the verified artifact before any recovery-path kill. Retained failure + // or an unknown result cannot silently fall back to a cold Environment. + next, err := r.saveCompute(ctx, owner, "suspending", state, owner.ComputeRetainedUntil) + if err != nil { + return err + } + owner = next + if validationErr != nil { + // Native absence alone does not settle capture or publish the archive. + return sandbox.ErrComputeUnconfirmed + } + _, err = r.saveCompute(ctx, next, "suspended", state, next.ComputeRetainedUntil) return err } -func (r *runtimeLifecycle) restoreIdleCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute) error { +func (r *runtimeLifecycle) restoreIdleCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute) (err error) { + defer func() { err = withObservationOwner(owner, err) }() activity, err := r.reader.Activity(ctx, owner.ID) if err != nil { return err @@ -223,24 +247,46 @@ func (r *runtimeLifecycle) restoreIdleCompute(ctx context.Context, p sandbox.San if state.Retained == nil || state.Target != nil { return sandbox.ErrOwnership } - target, err := p.NewCompute(ctx, runtimeReference(owner), state.Current.Generation+1, state.Retained) + state.RestoreID = uuid.NewString() + raw, err := json.Marshal(state) if err != nil { return err } - if target.Name == "" || target.Generation != state.Current.Generation+1 || target.RestoredFrom == nil || *target.RestoredFrom != *state.Retained { - return sandbox.ErrOwnership - } - state.Target, state.RestoreID = &target, uuid.NewString() - next, err := r.saveCompute(ctx, owner, "restoring", state, owner.ComputeRetainedUntil) + next, err := r.deployment.BeginRestore(ctx, owner, state.Retained.Compatibility, raw) if err != nil { return err } + if next.NodeID != r.nodeID { + // The durable route and reservation belong to the target node's lane. + if r.hintDestination != nil { + r.hintDestination(next.NodeID) + } + return nil + } return r.restoreCompute(ctx, p, next, state, false) } -func (r *runtimeLifecycle) restoreCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute, observeOnly bool) error { - if state.Target == nil || state.Retained == nil || state.Rollback { +func (r *runtimeLifecycle) restoreCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute, observeOnly bool) (err error) { + defer func() { err = withObservationOwner(owner, err) }() + if state.Retained == nil || state.Rollback || state.RestoreID == "" { return sandbox.ErrOwnership } + if state.Target == nil { + // NewCompute derives identity without starting native execution. A crash + // before this commit can safely repeat derivation on the reserved target. + target, err := p.NewCompute(ctx, runtimeReference(owner), state.Current.Generation+1, state.Retained) + if err != nil { + return err + } + if target.Name == "" || target.Generation != state.Current.Generation+1 || target.RestoredFrom == nil || *target.RestoredFrom != *state.Retained { + return sandbox.ErrOwnership + } + state.Target = &target + next, err := r.saveCompute(ctx, owner, "restoring", state, owner.ComputeRetainedUntil) + if err != nil { + return err + } + owner, observeOnly = next, false + } workspace, err := r.restoreWorkspace(ctx, owner) if err != nil { return err @@ -251,6 +297,18 @@ func (r *runtimeLifecycle) restoreCompute(ctx context.Context, p sandbox.Sandbox if err != nil { return err } + if result.RestoreAttemptClosed != "" { + if !result.ClosesRestoreAttempt(sandbox.ResumeRequest{OperationID: state.RestoreID, Retained: *state.Retained, Target: *state.Target, ReconcileOnly: observeOnly}) { + return sandbox.ErrOwnership + } + // Only an exact, durably closed attempt with no native dispatch can be replaced. + state.RestoreID = uuid.NewString() + next, err := r.saveCompute(ctx, owner, "restoring", state, owner.ComputeRetainedUntil) + if err != nil { + return err + } + return r.restoreCompute(ctx, p, next, state, false) + } if result.Status != "running" || !result.BootstrapComplete || sandbox.ValidateComputeResult(*state.Target, result.Compute) != nil { return sandbox.ErrComputeUnconfirmed } @@ -312,24 +370,10 @@ func (r *runtimeLifecycle) computeCapacity(ctx context.Context, key string) erro // restoreWorkspace follows the allocation's immutable generation, never the // current deployment's filesystem selection or sizing. func (r *runtimeLifecycle) restoreWorkspace(ctx context.Context, owner deployment.Allocation) (*workspacefs.Binding, error) { - record, err := r.reader.Allocation(ctx, runtimeReference(owner)) + spec, err := r.workspaceSpecification(ctx, owner.Key(), owner.ID) if err != nil { return nil, err } - if record.ID != owner.ID || record.InstallationID != owner.ProviderKey || record.Generation != owner.DeploymentGeneration || record.Released { - return nil, sandbox.ErrOwnership - } - raw := record.Deployment.Specification - if record.Generation != record.Deployment.Generation { - if record.Retained == nil || record.Retained.Generation != record.Generation { - return nil, sandbox.ErrOwnership - } - raw = record.Retained.Specification - } - var spec sandbox.DeploymentSpec - if err := json.Unmarshal(raw, &spec); err != nil { - return nil, sandbox.ErrInvalid - } if spec.Workspace == nil { return nil, nil } diff --git a/services/core/internal/execution/runtime_compute_compatibility_test.go b/services/core/internal/execution/runtime_compute_compatibility_test.go index 14c7c814e..07b38564e 100644 --- a/services/core/internal/execution/runtime_compute_compatibility_test.go +++ b/services/core/internal/execution/runtime_compute_compatibility_test.go @@ -1,13 +1,13 @@ package execution import ( + "context" "encoding/json" -"context" -"fmt" -"errors" -"github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" -"github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" -"github.com/MiniMax-AI/OpenAgentCore/services/core/internal/workspacefs" + "errors" + "fmt" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/workspacefs" "reflect" "testing" ) @@ -27,25 +27,56 @@ func TestBetaRetainedV1RoundTripPreservesEveryField(t *testing.T) { if json.Unmarshal([]byte(raw), &before) != nil || json.Unmarshal(out, &after) != nil || !reflect.DeepEqual(before, after) { t.Fatal("beta v1 field loss or rewrite") } - for _, invalid := range []string{`{}`, `{"protocol_version":"2"}`, `{"protocol_version":"1","snapshot":{}}`, raw + `{}`} { + for _, invalid := range []string{`{}`, `{"protocol_version":"2"}`, `{"protocol_version":"1","snapshot":{}}`, `{"protocol_version":"1","current":{"RestoredFrom":{"Digest":"legacy"}}}`, `{"protocol_version":"1","target":{"RestoredFrom":{"CheckpointRoot":"legacy"}}}`, raw + `{}`} { if decodeRuntimeCompute([]byte(invalid), new(runtimeCompute)) == nil { t.Fatal("incompatible receipt accepted") } } } +func TestReleasedDirectAllocationStillChecksCapacity(t *testing.T) { + for _, limit := range []string{"active", "retained"} { + t.Run(limit, func(t *testing.T) { + r := runtimeLifecycle{sessions: workspaceEnvironmentReader{}, config: RuntimeProvider{Mode: "direct", InstallationID: "fixture", Generation: 1, Suspension: &RuntimeSuspensionPolicy{MaxActive: 1, MaxRetained: 2}}, reader: &strictDeploymentReader{t: t, + environmentAllocation: func(context.Context, deployment.AllocationKey) (deployment.Allocation, error) { + return deployment.Allocation{State: "released"}, nil + }, + countComputeReservations: func(context.Context, string) (int64, error) { + if limit == "active" { + return 1, nil + } + return 0, nil + }, + countRetainedAllocations: func(context.Context, string) (int64, error) { return 2, nil }, + }} + if _, err := r.provision(t.Context(), "tenant", "environment", "fixture"); !errors.Is(err, ErrExecutionUnavailable) { + t.Fatal("released owner bypassed direct capacity", err) + } + }) + } +} + func TestRestoreUsesAllocationGenerationWorkspaceAfterDeploymentChange(t *testing.T) { - owner := deployment.Allocation{ID:"allocation",ProviderKey:"installation",DeploymentGeneration:4} - for _, originalExternal := range []bool{false,true} { - t.Run(fmt.Sprint(originalExternal),func(t *testing.T){ - original,current := sandbox.DeploymentSpec{},sandbox.DeploymentSpec{Workspace:&workspacefs.Declaration{}} - if originalExternal { original,current=current,original } - prior,_:=json.Marshal(original);latest,_:=json.Marshal(current) - r:=runtimeLifecycle{config:RuntimeProvider{Workspace:current.Workspace},reader:&strictDeploymentReader{t:t,allocation:func(context.Context,sandbox.Reference)(deployment.AllocationRecord,error){ - return deployment.AllocationRecord{ID:owner.ID,InstallationID:owner.ProviderKey,Generation:4,Deployment:deployment.Record{Generation:5,Specification:latest},Retained:&deployment.GenerationRecord{Generation:4,Specification:prior}},nil - }}} - binding,err:=r.restoreWorkspace(t.Context(),owner) - if originalExternal { if !errors.Is(err,workspacefs.ErrUnavailable){t.Fatal("original external storage was bypassed",err)} } else if err!=nil || binding!=nil {t.Fatal("old owned disk depended on new workspace",err)} - }) - } + owner := deployment.Allocation{ID: "allocation", ProviderKey: "installation", DeploymentGeneration: 4} + for _, originalExternal := range []bool{false, true} { + t.Run(fmt.Sprint(originalExternal), func(t *testing.T) { + original, current := sandbox.DeploymentSpec{}, sandbox.DeploymentSpec{Workspace: &workspacefs.Declaration{}} + if originalExternal { + original, current = current, original + } + prior, _ := json.Marshal(original) + r := runtimeLifecycle{config: RuntimeProvider{Workspace: current.Workspace}, reader: &strictDeploymentReader{t: t, lifecyclePlacement: func(context.Context, deployment.AllocationKey) (deployment.LifecyclePlacement, error) { + return deployment.LifecyclePlacement{AllocationID: owner.ID, Specification: prior}, nil + }}} + + binding, err := r.restoreWorkspace(t.Context(), owner) + if originalExternal { + if !errors.Is(err, workspacefs.ErrUnavailable) { + t.Fatal("original external storage was bypassed", err) + } + } else if err != nil || binding != nil { + t.Fatal("old owned disk depended on new workspace", err) + } + }) + } } diff --git a/services/core/internal/execution/runtime_compute_wake.go b/services/core/internal/execution/runtime_compute_wake.go index 3d485e21d..d27af6fd3 100644 --- a/services/core/internal/execution/runtime_compute_wake.go +++ b/services/core/internal/execution/runtime_compute_wake.go @@ -8,12 +8,14 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" "github.com/MiniMax-AI/OpenAgentCore/internal/runtimebootstrap" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/runtimegateway" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" ) -func (r *runtimeLifecycle) wakeCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute) error { +func (r *runtimeLifecycle) wakeCompute(ctx context.Context, p sandbox.SandboxProvider, owner deployment.Allocation, state runtimeCompute) (err error) { + defer func() { err = withObservationOwner(owner, err) }() if state.Rollback { if _, err := p.ResumeCompute(ctx, runtimeReference(owner), state.Current); err != nil { return err @@ -65,7 +67,9 @@ func (r *runtimeLifecycle) wakeCompute(ctx context.Context, p sandbox.SandboxPro if err != nil { return err } - if err := r.deployment.ClearWake(ctx, next, owner.ComputeActivityAt); err != nil { + observedActivity := owner.ComputeActivityAt + owner = next + if err := r.deployment.ClearWake(ctx, next, observedActivity); err != nil { return err } return r.observeConnection(ctx, next) @@ -75,6 +79,19 @@ func (r *runtimeLifecycle) cleanupCompute(ctx context.Context, p sandbox.Sandbox if err := r.lease.CheckOwnership(ctx); err != nil { return err } + if owner.ComputePhase == "suspended" && state.Retained != nil && owner.NodeID != "" { + next, err := r.deployment.RelocateCleanup(ctx, owner, state.Retained.Compatibility) + if err != nil { + return err + } + if next.NodeID != r.nodeID { + if r.hintDestination != nil { + r.hintDestination(next.NodeID) + } + return nil + } + owner = next + } // An uncommitted artifact is found by its persisted attempt, never a directory // glob. The helper's allocation lock also waits for an earlier unknown call. if owner.ComputePhase == "suspending" && state.Retained == nil { @@ -91,8 +108,10 @@ func (r *runtimeLifecycle) cleanupCompute(ctx context.Context, p sandbox.Sandbox return err } } - if err := ignoreComputeAbsent(p.KillCompute(ctx, runtimeReference(owner), state.Current)); err != nil { - return err + if owner.ComputePhase != "suspended" && owner.ComputePhase != "restoring" { + if err := ignoreComputeAbsent(p.KillCompute(ctx, runtimeReference(owner), state.Current)); err != nil { + return err + } } if state.Retained != nil { if err := ignoreComputeAbsent(p.DeleteRetained(ctx, runtimeReference(owner), *state.Retained)); err != nil { @@ -105,43 +124,98 @@ func (r *runtimeLifecycle) cleanupCompute(ctx context.Context, p sandbox.Sandbox // waitRuntimeAwake is called only for live Environment file operations, before // entering the Worker's work queues. Persisted history/artifact reads bypass it. -func (w *Worker) waitRuntimeAwake(ctx context.Context, environment sessions.Environment) error { +func (w *Worker) waitRuntimeAwake(ctx context.Context, environment sessions.Environment) (err error) { + // A deadline can arrive inside a read as well as between polling ticks. + // Both paths report the same unavailable outcome to the live file caller. + defer func() { + if ctx.Err() != nil && (errors.Is(err, context.Canceled) || errors.Is(err, context.DeadlineExceeded)) { + err = ErrExecutionUnavailable + } + }() key := deployment.AllocationKey{TenantID: environment.TenantID, EnvironmentID: environment.ID} - owner, err := w.dispatcher.DeploymentReader.EnvironmentAllocation(ctx, key) - if errors.Is(err, deployment.ErrNotFound) { - return nil - } - if err != nil { - return err - } - if owner.SessionDeleted || owner.Expired || owner.State == "cleanup_pending" || owner.State == "released" { - return ErrExecutionUnavailable - } - if owner.ComputePhase == "disabled" { - return nil - } - if err := w.dispatcher.Deployment.TouchActivity(ctx, environment.TenantID, environment.ID); err != nil { - return err - } - timer := time.NewTicker(100 * time.Millisecond) - defer timer.Stop() + ticker := time.NewTicker(250 * time.Millisecond) + defer ticker.Stop() + touched := "" for { - owner, err = w.dispatcher.DeploymentReader.EnvironmentAllocation(ctx, key) + owner, err := w.dispatcher.DeploymentReader.EnvironmentAllocation(ctx, key) + if errors.Is(err, deployment.ErrNotFound) { + return nil + } if err != nil { return err } - if owner.SessionDeleted || owner.Expired || owner.State != "running" { + if owner.SessionDeleted { return ErrExecutionUnavailable } - if owner.ComputePhase == "running" { - return nil + if owner.State == "released" || owner.State == "cleanup_pending" || owner.Expired { + // Public archive and deletion remain terminal. A qualified compute-only + // expiry leaves the same Environment live while the old writer is fenced. + current, err := w.dispatcher.SessionsReader.GetEnvironment(ctx, environment.TenantID, environment.ID) + if err != nil { + return err + } + if current.Status == "failed" || current.Status == "expired" { + return ErrExecutionUnavailable + } + if owner.State == "released" { + qualified, err := w.dispatcher.DeploymentReader.RetainedNativeHistory(ctx, key) + if err != nil { + return err + } + if !qualified { + return ErrExecutionUnavailable + } + // A qualified released allocation retains the existing wake intent. + // The common placement scan orders it with pending Session input. + if touched != owner.ID { + if err := w.dispatcher.Deployment.TouchActivity(ctx, environment.TenantID, environment.ID); err != nil { + return err + } + touched = owner.ID + w.wakeScheduler() + } + next, err := w.ProvisionEnvironment(ctx, environment.TenantID, environment.ID, owner.ProviderKey) + if err != nil && next.ID == "" && !errors.Is(err, placement.ErrNodeUnavailable) && !errors.Is(err, placement.ErrNodesPreparing) && !errors.Is(err, deployment.ErrAllocationConflict) { + return err + } + } + } else if owner.State == "running" && owner.CreateSettled && (owner.ComputePhase == "disabled" || owner.ComputePhase == "running") { + // Bootstrap completion precedes daemon registration and its first + // capability declaration. File work must wait for both receipts. + peer, peerErr := w.dispatcher.authorizedPeer(ctx, owner.DeviceID) + if peerErr == nil { + session, err := w.dispatcher.SessionsReader.GetSession(ctx, environment.TenantID, environment.SessionID) + if err != nil { + return err + } + kind, found, known := peer.AgentKindStatus(session.Engine) + if known { + if !found || !kind.Available { + return ErrExecutionUnavailable + } + return nil + } + } else if !errors.Is(peerErr, sessions.ErrNotFound) && !errors.Is(peerErr, runtimegateway.ErrDeviceNotRegistered) && !errors.Is(peerErr, runtimegateway.ErrSessionClosed) { + return peerErr + } + } else if owner.State == "running" && touched != owner.ID { + if err = w.dispatcher.Deployment.TouchActivity(ctx, environment.TenantID, environment.ID); err != nil { + return err + } + touched = owner.ID + if w.runtimes != nil { + select { + case w.runtimes.hints(owner.NodeID) <- struct{}{}: + default: + } + } } select { case <-ctx.Done(): return ErrExecutionUnavailable case <-w.stopped: return ErrExecutionUnavailable - case <-timer.C: + case <-ticker.C: } } } diff --git a/services/core/internal/execution/runtime_compute_wake_test.go b/services/core/internal/execution/runtime_compute_wake_test.go new file mode 100644 index 000000000..5e554262d --- /dev/null +++ b/services/core/internal/execution/runtime_compute_wake_test.go @@ -0,0 +1,56 @@ +package execution + +import ( + "context" + "errors" + "fmt" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" +) + +func TestRuntimeWakeMapsCancellationDuringAllocationRead(t *testing.T) { + storageError := errors.New("allocation read failed") + for _, test := range []struct { + name string + deadline bool + cancelBeforeResult bool + result error + want error + }{ + {name: "cancelled query", want: ErrExecutionUnavailable}, + {name: "deadline during query", deadline: true, want: ErrExecutionUnavailable}, + {name: "storage failure", result: storageError, want: storageError}, + {name: "storage failure concurrent with cancellation", cancelBeforeResult: true, result: storageError, want: storageError}, + } { + t.Run(test.name, func(t *testing.T) { + ctx, cancel := context.WithCancel(t.Context()) + if test.deadline { + cancel() + ctx, cancel = context.WithTimeout(t.Context(), time.Millisecond) + } + defer cancel() + called := false + reader := &strictDeploymentReader{t: t, environmentAllocation: func(query context.Context, _ deployment.AllocationKey) (deployment.Allocation, error) { + called = true + if test.cancelBeforeResult { + cancel() + } + if test.result != nil { + return deployment.Allocation{}, test.result + } + if !test.deadline { + cancel() + } + <-query.Done() + return deployment.Allocation{}, fmt.Errorf("timeout: %w", query.Err()) + }} + worker := &Worker{dispatcher: &Dispatcher{DeploymentReader: reader}, stopped: make(chan struct{})} + if err := worker.waitRuntimeAwake(ctx, sessions.Environment{}); !errors.Is(err, test.want) || !called { + t.Fatalf("allocation query outcome = %v, called = %v; want %v", err, called, test.want) + } + }) + } +} diff --git a/services/core/internal/execution/runtime_initialization.go b/services/core/internal/execution/runtime_initialization.go index b9a7ab23e..880175455 100644 --- a/services/core/internal/execution/runtime_initialization.go +++ b/services/core/internal/execution/runtime_initialization.go @@ -8,6 +8,7 @@ import ( "time" "github.com/MiniMax-AI/OpenAgentCore/internal/agentcapabilities" + "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" "github.com/MiniMax-AI/OpenAgentCore/internal/obs/log" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/environmentconfig" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/runtimegateway" @@ -91,7 +92,7 @@ func (w *Worker) runEnvironmentInitializations(ctx context.Context) error { } func (w *Worker) initializeEnvironment(ctx context.Context, owner sessions.EnvironmentInitialization) { - operation, cancel := context.WithTimeout(ctx, 30*time.Minute) + operation, cancel := context.WithTimeout(ctx, time.Duration(proto.RuntimePrepareMaxBudgetMS)*time.Millisecond) defer cancel() failure := sessions.ProvisioningFailure{} initializeAt := time.Now() @@ -128,14 +129,24 @@ func (w *Worker) prepareEnvironment(ctx context.Context, owner sessions.Environm if err != nil { return err } + placement, err := parseEnvironmentPlacement(environment.Configuration) + if err != nil { + return err + } + if placement.Type == "openai_hosted" { + info, found, known := peer.AgentKindStatus(owner.Engine) + if !known || !found || !info.Available || w.dispatcher.validateEnvironmentHistory(owner.Engine, environment, info.Capabilities.RetainedNativeHistory) != nil { + *failure = sessions.ProvisioningFailure{Step: sessions.ProvisioningHarness} + return sessions.ErrInvalidInput + } + } identity := agentcapabilities.Identity{EnvironmentID: owner.EnvironmentID, SessionID: owner.SessionID} operations := setupOperations(setup) for index := 0; index < len(cfg.Files)+len(operations); index++ { - step, stop := context.WithTimeout(ctx, 2*time.Minute) - err = w.lease.CheckOwnership(step) + err = w.lease.CheckOwnership(ctx) if err == nil { var currentPeer = peer - currentPeer, err = w.dispatcher.authorizedPeer(step, owner.DeviceID) + currentPeer, err = w.dispatcher.authorizedPeer(ctx, owner.DeviceID) if err == nil && currentPeer != peer { err = errors.New("Runtime connection changed during initialization") } @@ -144,16 +155,15 @@ func (w *Worker) prepareEnvironment(ctx context.Context, owner sessions.Environm if err == nil && index < len(cfg.Files) { var metadata environmentconfig.InitialFileMetadata var body []byte - metadata, body, err = w.dispatcher.SessionsReader.ReadInitialEnvironmentFile(step, owner.TenantID, owner.SessionID, index) + metadata, body, err = w.dispatcher.SessionsReader.ReadInitialEnvironmentFile(ctx, owner.TenantID, owner.SessionID, index) if err == nil { - err = installInitialFile(step, peer, identity, metadata, body) + err = installInitialFile(ctx, peer, identity, metadata, body) } } else if err == nil { command := operations[index-len(cfg.Files)] candidate = command.provisioningFailure(0) - err = runRuntimeSetup(step, peer, identity, command) + err = runRuntimeSetup(ctx, peer, identity, command) } - stop() if err != nil { var confirmed *runtimeStepFailure if errors.As(err, &confirmed) { diff --git a/services/core/internal/execution/runtime_lifecycle.go b/services/core/internal/execution/runtime_lifecycle.go index 44cd54de2..0916e70dc 100644 --- a/services/core/internal/execution/runtime_lifecycle.go +++ b/services/core/internal/execution/runtime_lifecycle.go @@ -4,6 +4,7 @@ import ( "context" "crypto/rand" "encoding/hex" + "encoding/json" "errors" "net/url" "sync" @@ -62,6 +63,7 @@ type runtimeLifecycle struct { pendingCursor string connections map[string]*runtimeConnection wakeHints chan struct{} + hintDestination func(string) } func newRuntimeManager(owner Owner, deployments *deployment.Service, deploymentReader deployment.Reader, sessionReader sessions.Reader, registry *runtimegateway.Registry, config *RuntimeProvider) (*runtimeManager, error) { @@ -73,7 +75,7 @@ func newRuntimeManager(owner Owner, deployments *deployment.Service, deploymentR return nil, sandbox.ErrInvalid } ctx, stop := context.WithCancel(context.Background()) - return &runtimeManager{workspaces: owner.Workspaces, workspaceGate: make(chan struct{}, 1), sessions: sessionReader, sessionExecution: owner.Sessions, deployment: owner.Deployment, deploymentService: deployments, deploymentReader: deploymentReader, lease: owner.Lease, registry: registry, setupInstallationID: config.InstallationID, loadDeployment: config.loadDeployment, prepareDeployment: config.prepareDeployment, publishUnconfigured: config.PublishUnconfigured, setupGate: make(chan struct{}, 1), mutationGate: make(chan struct{}, 1), ctx: ctx, cancel: stop, nodes: make(map[string]*runtimeNode), failed: make(chan error, 1), inventory: make(chan struct{}, 1)}, nil + return &runtimeManager{workspaces: owner.Workspaces, workspaceGate: make(chan struct{}, 1), sessions: sessionReader, sessionExecution: owner.Sessions, deployment: owner.Deployment, deploymentService: deployments, deploymentReader: deploymentReader, lease: owner.Lease, registry: registry, setupInstallationID: config.InstallationID, loadDeployment: config.loadDeployment, prepareDeployment: config.prepareDeployment, publishUnconfigured: config.PublishUnconfigured, setupGate: make(chan struct{}, 1), mutationGate: make(chan struct{}, 1), ctx: ctx, cancel: stop, nodes: make(map[string]*runtimeNode), failed: make(chan error, 1), inventory: make(chan struct{}, 1), placementWake: make(chan struct{}, 1)}, nil } func validatedRuntimeProvider(config *RuntimeProvider, registry *runtimegateway.Registry) (RuntimeProvider, error) { @@ -149,6 +151,17 @@ func (w *Worker) ProvisionEnvironment(ctx context.Context, tenant, environment, if !ready { return deployment.Allocation{}, ErrExecutionUnavailable } + key := deployment.AllocationKey{TenantID: tenant, EnvironmentID: environment} + existing, lookupErr := w.runtimes.deploymentReader.EnvironmentAllocation(ctx, key) + if lookupErr != nil && !errors.Is(lookupErr, deployment.ErrNotFound) { + return deployment.Allocation{}, lookupErr + } + if errors.Is(lookupErr, deployment.ErrNotFound) || existing.State == "released" { + // The common ordered inventory is the only placement entry point. + if _, err := w.runtimes.syncNodes(ctx); err != nil { + return deployment.Allocation{}, err + } + } nodeID, err := w.runtimes.deploymentService.LifecycleNode(ctx, tenant, environment) if err != nil { return deployment.Allocation{}, err @@ -184,7 +197,8 @@ func (r *runtimeLifecycle) provision(ctx context.Context, tenant, environment, p return deployment.Allocation{}, sandbox.ErrInvalid } key := deployment.AllocationKey{TenantID: tenant, EnvironmentID: environment} - if _, err := r.reader.EnvironmentAllocation(ctx, key); errors.Is(err, deployment.ErrNotFound) { + existing, lookupErr := r.reader.EnvironmentAllocation(ctx, key) + if errors.Is(lookupErr, deployment.ErrNotFound) || (lookupErr == nil && existing.State == "released") { if r.config.Generation == 0 { return deployment.Allocation{}, ErrExecutionUnavailable } @@ -200,26 +214,38 @@ func (r *runtimeLifecycle) provision(ctx context.Context, tenant, environment, p return deployment.Allocation{}, ErrExecutionUnavailable } } - } else if err != nil { + } else if lookupErr != nil { + return deployment.Allocation{}, lookupErr + } + spec, err := r.workspaceSpecification(ctx, deployment.AllocationKey{TenantID: tenant, EnvironmentID: environment}, "") + if err != nil { return deployment.Allocation{}, err } + if environmentValue.ExternalWorkspace && spec.Workspace == nil { + return deployment.Allocation{}, workspacefs.ErrUnsupported + } var workspace *workspacefs.Binding - if r.config.Workspace != nil { - if r.workspaces == nil { - return deployment.Allocation{}, workspacefs.ErrUnavailable - } + if spec.Workspace != nil { existing, lookupErr := r.reader.EnvironmentAllocation(ctx, deployment.AllocationKey{TenantID: tenant, EnvironmentID: environment}) if lookupErr != nil && !errors.Is(lookupErr, deployment.ErrNotFound) { return deployment.Allocation{}, lookupErr } if errors.Is(lookupErr, deployment.ErrNotFound) { - workspace, err = r.workspaces.Ensure(ctx, tenant, environment, r.config.WorkspaceRequirements, r.config.Resources.EnvironmentDiskMiB) + workspace, err = r.workspaces.Ensure(ctx, tenant, environment, r.config.WorkspaceRequirements, spec.Resources.EnvironmentDiskMiB) if err == nil && workspace == nil { err = workspacefs.ErrUnavailable } if err != nil { return deployment.Allocation{}, err } + } else if existing.State == "released" { + workspace, err = r.workspaces.GetReady(ctx, tenant, environment, r.config.WorkspaceRequirements, spec.Resources.EnvironmentDiskMiB) + if err != nil { + return deployment.Allocation{}, err + } + if workspace == nil { + return deployment.Allocation{}, workspacefs.ErrNotFound + } } else if existing.NodeID != r.nodeID { return existing, sandbox.ErrOwnership } @@ -467,3 +493,28 @@ func (r *runtimeLifecycle) computeFreshCapacity(ctx context.Context, key string) } return r.computeCapacity(ctx, key) } + +// workspaceSpecification reads the immutable generation selected for this +// Environment. A restore also fences the exact allocation before native I/O. +func (r *runtimeLifecycle) workspaceSpecification(ctx context.Context, key deployment.AllocationKey, allocationID string) (sandbox.DeploymentSpec, error) { + target, err := r.reader.LifecyclePlacement(ctx, key) + if err != nil { + return sandbox.DeploymentSpec{}, err + } + if allocationID != "" && target.AllocationID != allocationID { + return sandbox.DeploymentSpec{}, sandbox.ErrOwnership + } + var spec sandbox.DeploymentSpec + if err := json.Unmarshal(target.Specification, &spec); err != nil { + return sandbox.DeploymentSpec{}, sandbox.ErrInvalid + } + if spec.Workspace != nil { + if r.workspaces == nil || r.config.WorkspaceRequirements == nil { + return sandbox.DeploymentSpec{}, workspacefs.ErrUnavailable + } + if err := workspacefs.ValidateCombination(*r.config.WorkspaceRequirements, *spec.Workspace, spec.Resources.EnvironmentDiskMiB); err != nil { + return sandbox.DeploymentSpec{}, err + } + } + return spec, nil +} diff --git a/services/core/internal/execution/runtime_manager.go b/services/core/internal/execution/runtime_manager.go index 9eea4150f..5c3544e96 100644 --- a/services/core/internal/execution/runtime_manager.go +++ b/services/core/internal/execution/runtime_manager.go @@ -21,6 +21,8 @@ var errRuntimeTransition = fmt.Errorf("%w: sandbox configuration is changing", E type runtimeManager struct { workspaces *workspaces.ExecutionOperations workspaceCursor string + placementCursor deployment.PlacementDemandCursor + placementWake chan struct{} workspaceGate chan struct{} sessions sessions.Reader sessionExecution *sessions.ExecutionOperations @@ -96,7 +98,7 @@ func (m *runtimeManager) node(id string) (*runtimeNode, error) { deployment: m.deployment, deployments: m.deploymentService, reader: m.deploymentReader, lease: m.lease, registry: m.registry, config: m.config, nodeID: id, gate: make(chan struct{}, 1), ctx: ctx, stop: stop, - connections: make(map[string]*runtimeConnection), wakeHints: make(chan struct{}, 1), + connections: make(map[string]*runtimeConnection), wakeHints: make(chan struct{}, 1), hintDestination: m.hintDestination, }} m.nodes[id] = n } @@ -116,6 +118,19 @@ func (m *runtimeManager) node(id string) (*runtimeNode, error) { return n, nil } +// A committed handoff can introduce a node before the next inventory scan. +// Reuse its ordinary lane and coalesced wake; polling recovers missed hints. +func (m *runtimeManager) hintDestination(id string) { + n, err := m.node(id) + if err != nil { + return + } + select { + case n.lifecycle.wakeHints <- struct{}{}: + default: + } +} + func (m *runtimeManager) hints(id string) chan<- struct{} { m.mu.Lock() defer m.mu.Unlock() @@ -141,6 +156,9 @@ func (m *runtimeManager) syncNodes(ctx context.Context) ([]*runtimeNode, error) return nil, m.ctx.Err() } defer func() { <-m.inventory }() + if err := m.reservePlacements(ctx); err != nil { + return nil, err + } // A direct caller may add a newly registered node during the query. Only // entries present before this inventory snapshot can be retired by it. previous := m.snapshotNodes() @@ -268,6 +286,10 @@ func (m *runtimeManager) run(ctx context.Context) error { return m.ctx.Err() case err := <-m.failed: return err + case <-m.placementWake: + if _, err := m.syncNodes(ctx); err != nil && !errors.Is(err, errRuntimeTransition) { + return err + } case <-ticker.C: if err := m.deployment.CollectGenerations(ctx); err != nil { return err diff --git a/services/core/internal/execution/runtime_observation.go b/services/core/internal/execution/runtime_observation.go index e4d93e3d0..bc8cf2fab 100644 --- a/services/core/internal/execution/runtime_observation.go +++ b/services/core/internal/execution/runtime_observation.go @@ -8,7 +8,32 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" ) +// observationFailure retains the receipt held by the operation that failed. +// A later phase must never be queried and relabelled with an earlier failure. +type observationFailure struct { + owner deployment.Allocation + cause error +} + +func (e *observationFailure) Error() string { return e.cause.Error() } +func (e *observationFailure) Unwrap() error { return e.cause } + +func withObservationOwner(owner deployment.Allocation, err error) error { + if err == nil { + return nil + } + var observed *observationFailure + if errors.As(err, &observed) { + return err + } + return &observationFailure{owner: owner, cause: err} +} + func (r *runtimeLifecycle) recordObservation(ctx context.Context, owner deployment.Allocation, observed error) { + var failure *observationFailure + if errors.As(observed, &failure) { + owner = failure.owner + } if owner.NodeID == "" { return } diff --git a/services/core/internal/execution/runtime_observation_test.go b/services/core/internal/execution/runtime_observation_test.go new file mode 100644 index 000000000..aa5aecced --- /dev/null +++ b/services/core/internal/execution/runtime_observation_test.go @@ -0,0 +1,93 @@ +package execution + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +type observationCaptureProvider struct { + retentionProvider + captureError error +} + +func (p *observationCaptureProvider) Suspend(_ context.Context, q sandbox.SuspendRequest) (sandbox.ComputeState, error) { + if p.captureError != nil { + return sandbox.ComputeState{}, p.captureError + } + return sandbox.ComputeState{Compute: q.Source, Retained: &sandbox.RetainedState{Reference: "capture", Data: "opaque", OperationID: q.OperationID, SourceID: q.Source.ID, SourceName: q.Source.Name, SourceGeneration: q.Source.Generation, ID: "captured", Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}}, Status: "suspended", BootstrapComplete: true, SuspendSettled: true, ResourcesReleased: p.killError == nil}, nil +} + +func TestComputeFailureObservationUsesCommittedReceipt(t *testing.T) { + for _, stage := range []string{"capture", "source_cleanup"} { + t.Run(stage, func(t *testing.T) { + provider := &observationCaptureProvider{} + if stage == "capture" { + provider.captureError = sandbox.ErrComputeUnconfirmed + } else { + provider.killError = sandbox.ErrComputeUnconfirmed + } + fixture := workspaceSettlementFixtureWithSetup(t, provider, workspaceNodeSetup(t, true)) + r, session := fixture.lifecycle, fixture.session + initial, err := r.provision(t.Context(), session.TenantID, session.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + // Seed the acknowledged idle barrier; subsequent phase commits use the real + // Session-locked lifecycle operations, with real observation CAS writes. + if _, err = fixture.pool.Exec(t.Context(), `UPDATE environments SET initialization='complete',status='disconnected' WHERE id=$1`, initial.EnvironmentID); err != nil { + t.Fatal(err) + } + if _, err = fixture.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='quiescing',compute_revision=2,compute_retained_until=clock_timestamp()+interval '1 hour' WHERE id=$1`, initial.ID); err != nil { + t.Fatal(err) + } + original, err := r.reader.EnvironmentAllocation(t.Context(), initial.Key()) + if err != nil { + t.Fatal(err) + } + state := runtimeCompute{Version: sandbox.SuspensionStateVersion, Current: sandbox.Compute{ID: "source", Name: "source"}, SuspendID: "attempt"} + until := time.Now().Add(time.Hour) + suspending, err := r.saveCompute(t.Context(), original, "suspending", state, &until) + if err != nil { + t.Fatal(err) + } + failure := r.observeCompute(t.Context(), suspending) + if !errors.Is(failure, sandbox.ErrComputeUnconfirmed) { + t.Fatalf("capture failure = %v", failure) + } + // Model the outer reconcile entry retaining its earlier scan receipt. + r.recordObservation(t.Context(), original, failure) + current, err := r.reader.EnvironmentAllocation(t.Context(), initial.Key()) + if err != nil || current.ObservationError != "compute_unconfirmed" { + t.Fatalf("observation after transition = %q, %v", current.ObservationError, err) + } + wantRevision := int64(3) + if stage == "source_cleanup" { + wantRevision++ + } + if current.ComputeRevision != wantRevision { + t.Fatalf("revision = %d, want %d", current.ComputeRevision, wantRevision) + } + // A late result must not borrow a later operation's receipt or erase its + // diagnostic, even when both observations belong to the same allocation. + later, err := r.saveCompute(t.Context(), current, "waking", state, &until) + if err != nil { + t.Fatal(err) + } + r.recordObservation(t.Context(), later, sandbox.ErrOwnership) + r.recordObservation(t.Context(), original, failure) + current, err = r.reader.EnvironmentAllocation(t.Context(), initial.Key()) + if err != nil || current.ObservationError != "ownership_mismatch" { + t.Fatalf("late failure overwrote successor: %q, %v", current.ObservationError, err) + } + r.recordObservation(t.Context(), current, nil) + current, err = r.reader.EnvironmentAllocation(t.Context(), initial.Key()) + if err != nil || current.ObservationError != "" { + t.Fatalf("current successful observation did not clear diagnostic: %q, %v", current.ObservationError, err) + } + }) + } +} diff --git a/services/core/internal/execution/runtime_pending.go b/services/core/internal/execution/runtime_pending.go index 79bcc3a81..455fc5496 100644 --- a/services/core/internal/execution/runtime_pending.go +++ b/services/core/internal/execution/runtime_pending.go @@ -2,9 +2,12 @@ package execution import ( "context" + "errors" "time" "github.com/MiniMax-AI/OpenAgentCore/internal/obs/log" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" ) // provisionPending shares the existing lifecycle owner and serial gate. This @@ -35,3 +38,74 @@ func (r *runtimeLifecycle) provisionPending(ctx context.Context) error { } return nil } + +// reservePlacements runs under the inventory gate before node scans. Committed +// Sessions and inputs supply demand; a placement survives a disconnected caller. +// No native operation runs here or consumes a reservation for retained idle files. +func (m *runtimeManager) reservePlacements(ctx context.Context) error { + m.mu.Lock() + configuration := m.config + m.mu.Unlock() + if configuration.Mode != string(sandbox.DeploymentNodes) { + return nil + } + operation, cancel := context.WithTimeout(ctx, 5*time.Second) + defer cancel() + rows, next, err := m.deploymentReader.PlacementDemand(operation, m.placementCursor) + if err != nil { + // A bounded inventory read may time out while the execution owner is + // still healthy. Keep its cursor for the next hint or maintenance tick. + if errors.Is(err, context.DeadlineExceeded) && ctx.Err() == nil { + return m.lease.CheckOwnership(ctx) + } + return err + } + if len(rows) == 0 { + m.placementCursor = deployment.PlacementDemandCursor{} + return nil + } + for _, environment := range rows { + if operation.Err() != nil { + if ownership := m.lease.CheckOwnership(ctx); ownership != nil { + return ownership + } + return ctx.Err() + } + reserved, err := m.deployment.EnsurePlacement(operation, deployment.AllocationKey{TenantID: environment.TenantID, EnvironmentID: environment.ID}, configuration.InstallationID) + // A timed-out attempt yields its position, but later rows have not + // been attempted and must receive a fresh budget on the next scan. + m.placementCursor = deployment.PlacementDemandCursor{At: environment.At, EnvironmentID: environment.ID, Until: next.Until} + if err != nil { + if ownership := m.lease.CheckOwnership(ctx); ownership != nil { + return ownership + } + if ctx.Err() != nil { + return ctx.Err() + } + if operation.Err() != nil || errors.Is(err, context.DeadlineExceeded) { + return nil + } + log.Ctx(ctx).Warn("managed Runtime placement incomplete", "environment_id", environment.ID) + continue + } + node, err := m.node(reserved.NodeID) + if err != nil { + return err + } + select { + case node.lifecycle.wakeHints <- struct{}{}: + default: + } + if operation.Err() != nil { + if ownership := m.lease.CheckOwnership(ctx); ownership != nil { + return ownership + } + return ctx.Err() + } + } + m.placementCursor = next + if next.EnvironmentID == "" { + m.placementCursor = deployment.PlacementDemandCursor{} + } + return nil +} diff --git a/services/core/internal/execution/runtime_placement_demand_test.go b/services/core/internal/execution/runtime_placement_demand_test.go new file mode 100644 index 000000000..f4d5faa85 --- /dev/null +++ b/services/core/internal/execution/runtime_placement_demand_test.go @@ -0,0 +1,146 @@ +package execution + +import ( + "context" + "errors" + "fmt" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" + "github.com/google/uuid" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" +) + +func TestPlacementInventoryTimeoutRetainsOwnerAndCursor(t *testing.T) { + for _, tc := range []struct { + name string + failure error + fatal bool + }{ + {"bounded read", context.DeadlineExceeded, false}, + {"database failure", errors.New("database unavailable"), true}, + {"cancellation", context.Canceled, true}, + } { + t.Run(tc.name, func(t *testing.T) { + m := testRuntimeManager(t) + before := deployment.PlacementDemandCursor{EnvironmentID: "cursor"} + m.placementCursor = before + m.deploymentReader = &strictDeploymentReader{t: t, placementDemand: func(context.Context, deployment.PlacementDemandCursor) ([]deployment.PlacementDemand, deployment.PlacementDemandCursor, error) { + return nil, deployment.PlacementDemandCursor{}, tc.failure + }} + err := m.reservePlacements(t.Context()) + if (err != nil) != tc.fatal { + t.Fatal(err) + } + if m.placementCursor != before { + t.Fatal("failed page advanced cursor", m.placementCursor) + } + }) + } +} + +func TestPlacementInventoryTimeoutNeverMasksParentCancellationOrLostLease(t *testing.T) { + for _, canceled := range []bool{false, true} { + m := testRuntimeManager(t) + ctx, cancel := context.WithCancel(t.Context()) + defer cancel() + if canceled { + cancel() + } else { + m.lease = lostLease{} + } + m.deploymentReader = &strictDeploymentReader{t: t, placementDemand: func(context.Context, deployment.PlacementDemandCursor) ([]deployment.PlacementDemand, deployment.PlacementDemandCursor, error) { + return nil, deployment.PlacementDemandCursor{}, context.DeadlineExceeded + }} + if err := m.reservePlacements(ctx); err == nil { + t.Fatal("timeout masked cancellation or lost ownership") + } + } +} + +func TestPlacementReservationTimeoutYieldsOnlyAttemptedDemand(t *testing.T) { + for _, size := range []int{2, 32} { + t.Run(fmt.Sprint(size), func(t *testing.T) { + m := testRuntimeManager(t) + m.config.InstallationID = uuid.NewString() + horizon := time.Now().UTC() + rows := make([]deployment.PlacementDemand, size) + for i := range rows { + rows[i] = deployment.PlacementDemand{UnallocatedEnvironment: deployment.UnallocatedEnvironment{ID: uuid.NewString(), TenantID: uuid.NewString()}, At: horizon.Add(time.Duration(i-size) * time.Second)} + } + reads := 0 + m.deploymentReader = &strictDeploymentReader{t: t, placementDemand: func(ctx context.Context, after deployment.PlacementDemandCursor) ([]deployment.PlacementDemand, deployment.PlacementDemandCursor, error) { + reads++ + if ctx.Err() != nil { + t.Fatal("page read inherited expired budget", ctx.Err()) + } + if reads == 1 { + next := deployment.PlacementDemandCursor{Until: horizon} + if size == 32 { + next.At = rows[size-1].At + next.EnvironmentID = rows[size-1].ID + } + return rows, next, nil + } + want := deployment.PlacementDemandCursor{At: rows[0].At, EnvironmentID: rows[0].ID, Until: horizon} + if after != want { + t.Fatalf("later unattempted rows were skipped or horizon changed: got %#v, want %#v", after, want) + } + return rows[1:], deployment.PlacementDemandCursor{Until: horizon}, nil + }} + attempts := []string{} + _, m.deployment = deploymentOperations(t, &strictDeploymentStorage{t: t}, &strictDeploymentReader{t: t}, &strictExecutionStorage{t: t, withReservation: func(ctx context.Context, key deployment.AllocationKey, _ func(sessions.LockedSession, deployment.ReservationTx) error) error { + if ctx.Err() != nil { + t.Fatal("reservation attempted with spent context", ctx.Err()) + } + attempts = append(attempts, key.EnvironmentID) + if len(attempts) == 1 { + return context.DeadlineExceeded + } + return placement.ErrNodeUnavailable + }}) + if err := m.reservePlacements(t.Context()); err != nil { + t.Fatal(err) + } + if len(attempts) != 1 { + t.Fatal("spent reservation budget advanced later rows", attempts) + } + if err := m.reservePlacements(t.Context()); err != nil { + t.Fatal(err) + } + if len(attempts) != size { + t.Fatal("later rows did not get a fresh attempt", attempts) + } + for i, id := range attempts { + if id != rows[i].ID { + t.Fatal("attempt order changed", attempts) + } + } + if m.placementCursor != (deployment.PlacementDemandCursor{}) { + t.Fatal("completed short page did not wrap", m.placementCursor) + } + }) + } +} + +func TestPlacementBudgetEndsBeforeAttemptPreservesCursor(t *testing.T) { + m := testRuntimeManager(t) + before := deployment.PlacementDemandCursor{EnvironmentID: uuid.NewString(), At: time.Now().Add(-time.Minute), Until: time.Now()} + m.placementCursor = before + ctx, cancel := context.WithCancel(t.Context()) + defer cancel() + m.deploymentReader = &strictDeploymentReader{t: t, placementDemand: func(context.Context, deployment.PlacementDemandCursor) ([]deployment.PlacementDemand, deployment.PlacementDemandCursor, error) { + cancel() + return []deployment.PlacementDemand{{UnallocatedEnvironment: deployment.UnallocatedEnvironment{ID: uuid.NewString(), TenantID: uuid.NewString()}}}, deployment.PlacementDemandCursor{Until: before.Until}, nil + }} + // There is deliberately no deployment executor: no reservation may start + // after the read exhausts the parent budget. + if err := m.reservePlacements(ctx); !errors.Is(err, context.Canceled) { + t.Fatal(err) + } + if m.placementCursor != before { + t.Fatal("unattempted demand advanced cursor", m.placementCursor) + } +} diff --git a/services/core/internal/execution/runtime_replacement_test.go b/services/core/internal/execution/runtime_replacement_test.go new file mode 100644 index 000000000..6e5aa3212 --- /dev/null +++ b/services/core/internal/execution/runtime_replacement_test.go @@ -0,0 +1,510 @@ +package execution + +import ( + "context" + "encoding/json" + "errors" + "net/http" + "net/http/httptest" + "net/url" + "strings" + "testing" + "time" + + v1 "github.com/MiniMax-AI/OpenAgentCore/contracts/agents-api/v1" + "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" + "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto/prototest" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/adminaudit" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/identity" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/providercontract" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/runtimegateway" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/workspacefs" + "github.com/google/uuid" + "github.com/gorilla/websocket" +) + +type retentionProvider struct { + sandbox.SandboxProvider + killError error + creates, kills, snapshots int + bootstraps chan sandbox.Bootstrap +} + +func (p *retentionProvider) ProviderOperations() providercontract.Operations { + return microsandbox.Operations() +} +func (p *retentionProvider) Create(_ context.Context, q sandbox.Bootstrap) (sandbox.Info, error) { + p.creates++ + if p.bootstraps != nil { + p.bootstraps <- q + } + return sandbox.Info{Reference: q.Reference, ProviderID: q.AllocationID, State: "running", BootstrapComplete: true, CreateSettled: true}, nil +} +func (p *retentionProvider) KillCompute(context.Context, sandbox.Reference, sandbox.Compute) error { + p.kills++ + return p.killError +} +func (p *retentionProvider) DeleteRetained(context.Context, sandbox.Reference, sandbox.RetainedState) error { + p.snapshots++ + return nil +} + +func TestCapacityPressurePreservesSuspendedSnapshot(t *testing.T) { + provider := &retentionProvider{} + fixture := newWorkspaceSettlementFixture(t, provider) + r, session := fixture.lifecycle, fixture.session + owner, err := r.provision(t.Context(), session.TenantID, session.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + for _, statement := range []struct { + sql string + arg any + }{ + {`UPDATE runtime_nodes SET max_active=1,max_retained=1 WHERE id=$1`, owner.NodeID}, + {`UPDATE environments SET initialization='complete',status='disconnected' WHERE id=$1`, owner.EnvironmentID}, + {`UPDATE devices SET supported_agent_kinds='[{"kind":"codex","available":true,"capabilities":{"retained_native_history":true}}]' WHERE id=$1`, owner.DeviceID}, + {`UPDATE runtime_allocations SET state='running',compute_phase='suspended',compute_state='{"protocol_version":"1","current":{"ID":"old-compute"},"retained":{"ID":"old-snapshot","Data":"opaque","Compatibility":{"artifact_domain":"fixture-store","execution_class":"fixture-runtime"}}}',compute_retained_until=clock_timestamp()+interval '24 hours' WHERE id=$1`, owner.ID}, + } { + if _, err := fixture.pool.Exec(t.Context(), statement.sql, statement.arg); err != nil { + t.Fatal(err) + } + } + waiting, err := fixture.sessions.CreateSession(t.Context(), session.TenantID, sessions.CreateSession{SupportsRetainedNativeHistory: true, Creator: identity.Subject{Kind: "service_account", ID: "fixture"}, Engine: "codex", IdempotencyKey: uuid.NewString(), Configuration: json.RawMessage(`{"agent":{"model":"test-model"},"environment":{"type":"openai_hosted","network":{"access":"disabled"}}}`), ModelProvider: &v1.ModelProviderInput{Protocol: "responses", BaseURL: "https://model.fixture.example/v1", APIKey: "fixture-key"}, ModelProviderSource: v1.ExecutionSourceSession}) + if err != nil { + t.Fatal(err) + } + input, err := fixture.sessions.ReserveEnvironmentInput(t.Context(), session.TenantID, waiting.Session.ID, "waiting-for-capacity", []sessions.Input{{Kind: "message", Payload: json.RawMessage(`{"input":[{"role":"user","content":[{"type":"input_text","text":"continue"}]}]}`)}}) + if err != nil { + t.Fatal(err) + } + owner, err = r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + for range 3 { + if err := r.observeCompute(t.Context(), owner); err != nil { + t.Fatal(err) + } + current, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || current.Expired || current.ComputePhase != "suspended" || current.ComputeRevision != owner.ComputeRevision || current.ComputeRetainedUntil == nil || !current.ComputeRetainedUntil.Equal(*owner.ComputeRetainedUntil) { + t.Fatal("pressure changed retained snapshot", current, err) + } + if _, err := r.deployment.EnsurePlacement(t.Context(), deployment.AllocationKey{TenantID: session.TenantID, EnvironmentID: waiting.Session.Environment.ID}, r.config.InstallationID); !errors.Is(err, placement.ErrNodeUnavailable) { + t.Fatal("released retained capacity without cleanup", err) + } + } + pending, err := r.sessions.GetEnvironmentInputReservation(t.Context(), session.TenantID, waiting.Session.ID, input.ID) + if err != nil || pending.State != sessions.EnvironmentInputPending || !pending.Deadline.Equal(input.Deadline) { + t.Fatal("pressure changed input deadline", pending, err) + } + + if provider.kills != 0 || provider.snapshots != 0 || provider.creates != 1 { + t.Fatal("pressure changed native resources", provider) + } +} + +func TestExpiredRetainedComputePreservesSessionAndPendingInput(t *testing.T) { + provider := &retentionProvider{killError: sandbox.ErrComputeUnconfirmed} + fixture := newWorkspaceSettlementFixture(t, provider) + r, session := fixture.lifecycle, fixture.session + owner, err := r.provision(t.Context(), session.TenantID, session.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + binding, err := fixture.storage.Get(t.Context(), session.TenantID, session.Environment.ID) + if err != nil { + t.Fatal(err) + } + // Model the committed completion and suspension receipts. Provider callbacks + // remain deterministic fixtures; these records do not claim native evidence. + if _, err = fixture.pool.Exec(t.Context(), `UPDATE environments SET initialization='complete',status='disconnected' WHERE id=$1`, owner.EnvironmentID); err != nil { + t.Fatal(err) + } + if _, err = fixture.pool.Exec(t.Context(), `UPDATE devices SET supported_agent_kinds='[{"kind":"codex","available":true,"capabilities":{"retained_native_history":true}}]' WHERE id=$1`, owner.DeviceID); err != nil { + t.Fatal(err) + } + if _, err = fixture.pool.Exec(t.Context(), `UPDATE session_devices SET native_session_id='retained-native-session' WHERE device_id=$1`, owner.DeviceID); err != nil { + t.Fatal(err) + } + compute := runtimeCompute{Version: sandbox.SuspensionStateVersion, Current: sandbox.Compute{ID: "old-compute"}, Retained: &sandbox.RetainedState{ID: "old-snapshot", Data: "native-state"}} + raw, _ := json.Marshal(compute) + if _, err = fixture.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspending',compute_state=$2,compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID, raw); err != nil { + t.Fatal(err) + } + input, err := fixture.sessions.ReserveEnvironmentInput(t.Context(), session.TenantID, session.ID, "after-retention", []sessions.Input{{Kind: "message", Payload: json.RawMessage(`{"input":[{"role":"user","content":[{"type":"input_text","text":"resume"}]}]}`)}}) + if err != nil { + t.Fatal(err) + } + for _, settled := range []bool{false, true} { + if settled { + provider.killError = nil + } + owner, err = r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + err = r.observe(t.Context(), owner) + if !settled && !errors.Is(err, sandbox.ErrComputeUnconfirmed) { + t.Fatal("unknown old writer was settled", err) + } + if settled && err != nil { + t.Fatal(err) + } + current, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + want := "cleanup_pending" + if settled { + want = "released" + } + if current.State != want || current.ID != owner.ID || !current.CreateSettled { + t.Fatal("compute ownership changed without settlement", current) + } + environment, err := r.sessions.GetEnvironment(t.Context(), session.TenantID, owner.EnvironmentID) + if err != nil || environment.Status != "disconnected" || environment.Initialization != "complete" { + t.Fatal("compute expiry terminated retained Environment", environment, err) + } + pending, err := r.sessions.GetEnvironmentInputReservation(t.Context(), session.TenantID, session.ID, input.ID) + if err != nil || pending.State != sessions.EnvironmentInputPending { + t.Fatal("compute expiry consumed pending input", pending, err) + } + var nativeID string + if err = fixture.pool.QueryRow(t.Context(), `SELECT native_session_id FROM session_devices WHERE session_id=$1`, session.ID).Scan(&nativeID); err != nil || nativeID != "retained-native-session" { + t.Fatal("compute expiry discarded native identity", nativeID, err) + } + if _, err = r.workspaces.DeleteBatch(t.Context(), ""); err != nil { + t.Fatal(err) + } + retained, err := fixture.storage.Get(t.Context(), session.TenantID, owner.EnvironmentID) + if err != nil || retained.Reference != binding.Reference || retained.Configuration.ID != binding.Configuration.ID || fixture.control.deletes != 0 { + t.Fatal("compute expiry changed retained filesystem", retained, err) + } + } + if provider.creates != 1 || provider.snapshots != 1 { + t.Fatal("expiry retried creation or deleted an unconfirmed writer snapshot", provider.creates, provider.snapshots) + } +} + +func TestRetainedEnvironmentDemandRecreatesCompute(t *testing.T) { + for _, demand := range []string{"input", "live_file"} { + t.Run(demand, func(t *testing.T) { + provider := &retentionProvider{} + fixture := workspaceSettlementFixtureWithSetup(t, provider, workspaceNodeSetup(t, true)) + r, session := fixture.lifecycle, fixture.session + owner, err := r.provision(t.Context(), session.TenantID, session.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + binding, err := fixture.storage.Get(t.Context(), session.TenantID, owner.EnvironmentID) + if err != nil { + t.Fatal(err) + } + for _, statement := range []struct { + q string + arg any + }{ + {`UPDATE environments SET initialization='complete',status='disconnected' WHERE id=$1`, owner.EnvironmentID}, + {`UPDATE devices SET supported_agent_kinds='[{"kind":"codex","available":true,"capabilities":{"retained_native_history":true}}]' WHERE id=$1`, owner.DeviceID}, + {`UPDATE session_devices SET native_session_id='retained-native-session' WHERE device_id=$1`, owner.DeviceID}, + {`UPDATE runtime_allocations SET compute_phase='suspended',compute_state='{"protocol_version":"1","current":{"ID":"old-compute"},"retained":{"ID":"old-snapshot","Data":"native-state","Compatibility":{"artifact_domain":"fixture-store","execution_class":"fixture-runtime"}}}',compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID}, + } { + if _, err = fixture.pool.Exec(t.Context(), statement.q, statement.arg); err != nil { + t.Fatal(err) + } + } + owner, err = r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + if err = r.observe(t.Context(), owner); err != nil { + t.Fatal(err) + } + owner, err = r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || owner.State != "released" { + t.Fatal(owner, err) + } + r.config.Mode = string(sandbox.DeploymentNodes) + manager := &runtimeManager{ctx: t.Context(), config: r.config, setupGate: make(chan struct{}, 1), inventory: make(chan struct{}, 1), nodes: make(map[string]*runtimeNode), workspaces: r.workspaces, sessions: r.sessions, sessionExecution: r.sessionExecution, deployment: r.deployment, deploymentService: r.deployments, deploymentReader: r.reader, lease: r.lease, registry: r.registry} + // Neither retained storage nor an old receipt is demand by itself. + if err = manager.reservePlacements(t.Context()); err != nil { + t.Fatal(err) + } + candidates, err := r.reader.UnallocatedEnvironments(t.Context(), r.nodeID, "") + if err != nil || len(candidates) != 0 { + t.Fatal("idle retained Environment reserved compute", candidates, err) + } + if demand == "input" { + _, err = fixture.sessions.ReserveEnvironmentInput(t.Context(), session.TenantID, session.ID, "replacement-input", []sessions.Input{{Kind: "message", Payload: json.RawMessage(`{"input":[{"role":"user","content":[{"type":"input_text","text":"resume"}]}]}`)}}) + if err != nil { + t.Fatal(err) + } + if err = manager.reservePlacements(t.Context()); err != nil { + t.Fatal(err) + } + if err = r.provisionPending(t.Context()); err != nil { + t.Fatal(err) + } + } else { + worker := &Worker{runtimes: manager, dispatcher: &Dispatcher{Registry: r.registry, DeploymentReader: r.reader, Deployment: r.deployments, SessionsReader: r.sessions}, stopped: make(chan struct{}), directoryReads: make(chan directoryReadRequest)} + provider.bootstraps = make(chan sandbox.Bootstrap, 1) + ctx, cancel := context.WithTimeout(t.Context(), 5*time.Second) + defer cancel() + completed := make(chan error, 1) + go func() { _, err := worker.ReadEnvironmentDirectory(ctx, *session.Environment, ""); completed <- err }() + var bootstrap sandbox.Bootstrap + select { + case bootstrap = <-provider.bootstraps: + case <-ctx.Done(): + t.Fatal(ctx.Err()) + } + assertWaiting := func() { + t.Helper() + select { + case err := <-completed: + t.Fatal("file call returned before Runtime readiness", err) + case <-worker.directoryReads: + t.Fatal("file work queued before Runtime readiness") + case <-time.After(300 * time.Millisecond): + } + } + assertWaiting() + handler := runtimegateway.NewHandler(runtimegateway.HandlerConfig{Authenticator: runtimegateway.NewAuthenticator(r.sessions), Registry: r.registry}) + server := httptest.NewServer(http.HandlerFunc(handler.WS)) + defer server.Close() + query := url.Values{"device_id": {bootstrap.DeviceID}, "version": {proto.Version}} + conn, _, err := websocket.DefaultDialer.Dial("ws"+strings.TrimPrefix(server.URL, "http")+"?"+query.Encode(), http.Header{"Authorization": {"Bearer " + bootstrap.Credential}}) + if err != nil { + t.Fatal(err) + } + defer conn.Close() + assertWaiting() + heartbeat, _ := proto.NewEnvelope(proto.TypeHeartbeat, "", proto.HeartbeatPayload{SupportedAgentKinds: []proto.SupportedAgentKind{{Kind: "codex", Available: true, Capabilities: prototest.Capabilities(proto.AgentKindCapabilities{LocalEnvironment: proto.CapabilitySupported, Preparation: proto.CapabilitySupported, WorkspaceReadPreparation: proto.CapabilitySupported, RetainedNativeHistory: proto.CapabilitySupported})}}}) + if err = conn.WriteJSON(heartbeat); err != nil { + t.Fatal(err) + } + // The public file entry point may enqueue only after the actual + // credential-authorized connection has advertised its Harness. + select { + case request := <-worker.directoryReads: + request.reply(directoryReadResult{directory: proto.WorkspaceDirectoryResult{Entries: []proto.WorkspaceDirectoryEntry{}}}) + case err := <-completed: + t.Fatal("file call failed", err) + case <-ctx.Done(): + t.Fatal(ctx.Err()) + } + if err = <-completed; err != nil { + t.Fatal(err) + } + } + replacement, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || replacement.ID == owner.ID || replacement.DeviceID == owner.DeviceID || replacement.State != "running" || !replacement.CreateSettled { + t.Fatal("demand failed to create fresh owner", replacement, err) + } + environment, err := r.sessions.GetEnvironment(t.Context(), session.TenantID, owner.EnvironmentID) + if err != nil || environment.Initialization != "complete" { + t.Fatal("replacement replayed initialization", environment, err) + } + retained, err := fixture.storage.Get(t.Context(), session.TenantID, owner.EnvironmentID) + if err != nil || retained.Reference != binding.Reference || retained.Configuration.ID != binding.Configuration.ID || provider.creates != 2 { + t.Fatal("replacement changed immutable filesystem or create count", retained, provider.creates, err) + } + var nativeID string + if err = fixture.pool.QueryRow(t.Context(), `SELECT native_session_id FROM session_devices WHERE session_id=$1`, session.ID).Scan(&nativeID); err != nil || nativeID != "retained-native-session" { + t.Fatal("replacement lost native history identity", nativeID, err) + } + }) + } +} + +func workspaceNodeSetup(t *testing.T, external bool) func(Owner, *deployment.Service, deployment.Reader) (string, string) { + return func(owner Owner, service *deployment.Service, reader deployment.Reader) (string, string) { + installation := uuid.NewString() + if err := owner.Deployment.Claim(t.Context(), installation); err != nil { + t.Fatal(err) + } + specification := sandbox.DeploymentSpec{Resources: sandbox.Resources{CPUs: 2, MemoryMiB: 2048, RootDiskMiB: 8192}, Workspace: &workspacefs.Declaration{Attachment: workspacefs.AttachmentHostDirectory, UserXAttr: true}, Runtime: &sandbox.RuntimeRelease{SourceCommit: strings.Repeat("a", 40), ImageID: "sha256:" + strings.Repeat("b", 64), ImageManifestDigest: "sha256:" + strings.Repeat("c", 64), MicrosandboxRef: "oac-runtime@sha256:" + strings.Repeat("d", 64), RuntimeSHA256: strings.Repeat("e", 64), FirmwareSHA256: strings.Repeat("f", 64)}} + if !external { + specification.Workspace = nil + specification.Resources.EnvironmentDiskMiB = 8192 + } + view, err := owner.Deployment.Initialize(t.Context(), installation, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: specification}) + if err != nil { + t.Fatal(err) + } + token, err := service.CreateEnrollment(t.Context(), deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + if err != nil { + t.Fatal(err) + } + node := deployment.Enrollment{NodeID: uuid.NewString(), Name: "replacement node", Credential: strings.Repeat("n", 64), Provider: view.Provider, BackendFingerprint: strings.Repeat("b", 64), DeploymentGeneration: view.Generation, SpecificationDigest: view.SpecificationDigest, CoreURL: fixturePublicURL} + if _, err = service.Enroll(t.Context(), token.Token, node); err != nil { + t.Fatal(err) + } + epoch, err := reader.OwnerEpoch(t.Context()) + if err != nil { + t.Fatal(err) + } + connection := uuid.NewString() + if err = service.ConnectNode(t.Context(), node.NodeID, connection, epoch); err != nil { + t.Fatal(err) + } + if err = service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}, []sandbox.GenerationStatus{{Generation: view.Generation, SpecificationDigest: view.SpecificationDigest, State: "ready", Checkpoint: &sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}}}); err != nil { + t.Fatal(err) + } + return installation, node.NodeID + + } +} + +func TestProvisionUsesReservedGenerationStorageDuringUpdate(t *testing.T) { + for _, external := range []bool{false, true} { + t.Run(map[bool]string{false: "owned_ready_external_target", true: "external_ready_owned_target"}[external], func(t *testing.T) { + provider := &retentionProvider{bootstraps: make(chan sandbox.Bootstrap, 1)} + f := workspaceSettlementFixtureWithSetup(t, provider, workspaceNodeSetup(t, external)) + r, s := f.lifecycle, f.session + setup, err := r.deployments.Setup(t.Context()) + if err != nil { + t.Fatal(err) + } + target := setup.Specification + target.Workspace = &workspacefs.Declaration{Attachment: workspacefs.AttachmentHostDirectory, UserXAttr: true} + target.Resources.EnvironmentDiskMiB = 0 + if external { + target.Workspace = nil + target.Resources.EnvironmentDiskMiB = 8192 + } + if _, err = r.deployment.Update(adminaudit.WithSource(t.Context(), adminaudit.Source{CredentialID: "fixture-admin", RequestID: uuid.NewString(), TraceID: uuid.NewString()}), r.config.InstallationID, sandbox.Selection{Provider: setup.Provider, ExpectedGeneration: setup.Generation, DeploymentSpec: target}); err != nil { + t.Fatal(err) + } + // Neither today's target nor the lane's cached storage mode selects storage. + r.config.Workspace = target.Workspace + owner, err := r.provision(t.Context(), s.TenantID, s.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + if owner.DeploymentGeneration != setup.Generation { + t.Fatal("lost reserved generation", owner) + } + bootstrap := <-provider.bootstraps + if (bootstrap.Workspace != nil) != external { + t.Fatal("storage mode followed target instead of reservation", external, bootstrap.Workspace) + } + environment, err := r.sessions.GetEnvironment(t.Context(), s.TenantID, s.Environment.ID) + if err != nil || environment.ExternalWorkspace != external { + t.Fatal("immutable binding projection", environment, err) + } + }) + } +} + +func TestProvisionRejectsRetainedBindingOnOwnedGeneration(t *testing.T) { + provider := &retentionProvider{bootstraps: make(chan sandbox.Bootstrap, 1)} + f := workspaceSettlementFixtureWithSetup(t, provider, workspaceNodeSetup(t, false)) + r, s := f.lifecycle, f.session + if _, err := r.workspaces.Ensure(t.Context(), s.TenantID, s.Environment.ID, r.config.WorkspaceRequirements, 0); err != nil { + t.Fatal(err) + } + if _, err := r.provision(t.Context(), s.TenantID, s.Environment.ID, r.config.InstallationID); !errors.Is(err, workspacefs.ErrUnsupported) { + t.Fatal("binding silently dropped", err) + } + if provider.creates != 0 { + t.Fatal("incompatible compute was created") + } + if _, err := r.reader.EnvironmentAllocation(t.Context(), deployment.AllocationKey{TenantID: s.TenantID, EnvironmentID: s.Environment.ID}); !errors.Is(err, deployment.ErrNotFound) { + t.Fatal("incompatible compute reserved", err) + } +} + +// A deterministic provider stops after capturing Resume, before subsequent +// wake operations. The allocation, generation and filesystem binding are real. +type resumeStorageProvider struct { + retentionProvider + request *sandbox.ResumeRequest + stop error +} + +func (p *resumeStorageProvider) Resume(_ context.Context, request sandbox.ResumeRequest) (sandbox.ComputeState, error) { + p.request = &request + return sandbox.ComputeState{}, p.stop +} + +func TestRestoreUsesAllocationGenerationStorageInsteadOfCachedLane(t *testing.T) { + for _, external := range []bool{false, true} { + t.Run(map[bool]string{false: "owned_allocation_external_lane", true: "external_allocation_owned_lane"}[external], func(t *testing.T) { + stop := errors.New("resume request captured") + provider := &resumeStorageProvider{stop: stop} + f := workspaceSettlementFixtureWithSetup(t, provider, workspaceNodeSetup(t, external)) + r, s := f.lifecycle, f.session + owner, err := r.provision(t.Context(), s.TenantID, s.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + setup, err := r.deployments.Setup(t.Context()) + if err != nil { + t.Fatal(err) + } + target := setup.Specification + target.Workspace = &workspacefs.Declaration{Attachment: workspacefs.AttachmentHostDirectory, UserXAttr: true} + target.Resources.EnvironmentDiskMiB = 0 + if external { + target.Workspace = nil + target.Resources.EnvironmentDiskMiB = 8192 + } + audit := adminaudit.WithSource(t.Context(), adminaudit.Source{CredentialID: "fixture-admin", RequestID: uuid.NewString(), TraceID: uuid.NewString()}) + if _, err = r.deployment.Update(audit, r.config.InstallationID, sandbox.Selection{Provider: setup.Provider, ExpectedGeneration: setup.Generation, DeploymentSpec: target}); err != nil { + t.Fatal(err) + } + r.config.Workspace = target.Workspace + r.config.Resources = target.Resources + state := runtimeCompute{Target: &sandbox.Compute{ID: "replacement-compute", Generation: 2}, Retained: &sandbox.RetainedState{ID: "snapshot", Data: "native-state"}, RestoreID: uuid.NewString()} + if err = r.restoreCompute(t.Context(), provider, owner, state, false); !errors.Is(err, stop) { + t.Fatal(err) + } + if provider.request == nil || (provider.request.Workspace != nil) != external { + t.Fatal("Resume binding followed cached lane", external, provider.request) + } + if external { + stored, err := f.storage.Get(t.Context(), s.TenantID, s.Environment.ID) + if err != nil || provider.request.Workspace.Attachment.Reference != stored.Reference { + t.Fatal("Resume lost retained object identity", err) + } + } + provider.request = nil + stale := owner + stale.ID = uuid.NewString() + if err = r.restoreCompute(t.Context(), provider, stale, state, false); !errors.Is(err, sandbox.ErrOwnership) || provider.request != nil { + t.Fatal("stale owner reached native Resume", err) + } + }) + } +} + +func TestSuspendedAllocationDemandDoesNotColdReplaceWriter(t *testing.T) { + for _, external := range []bool{false, true} { + t.Run(map[bool]string{false: "owned", true: "external"}[external], func(t *testing.T) { + provider := &retentionProvider{} + f := workspaceSettlementFixtureWithSetup(t, provider, workspaceNodeSetup(t, external)) + r, s := f.lifecycle, f.session + owner, err := r.provision(t.Context(), s.TenantID, s.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspended',compute_retained_until=clock_timestamp()+interval '1 hour' WHERE id=$1`, owner.ID); err != nil { + t.Fatal(err) + } + if _, err = f.sessions.ReserveEnvironmentInput(t.Context(), s.TenantID, s.ID, "owned-pause", []sessions.Input{{Kind: "message", Payload: json.RawMessage(`{"input":[{"role":"user","content":[{"type":"input_text","text":"resume"}]}]}`)}}); err != nil { + t.Fatal(err) + } + replay, err := r.provision(t.Context(), s.TenantID, s.Environment.ID, r.config.InstallationID) + if err != nil || replay.ID != owner.ID || replay.DeviceID != owner.DeviceID || !replay.Replayed || replay.ComputePhase != "suspended" || provider.creates != 1 || provider.kills != 0 { + t.Fatal("owned suspended writer was cold replaced", replay, err) + } + }) + } +} diff --git a/services/core/internal/execution/runtime_restore_attempt_test.go b/services/core/internal/execution/runtime_restore_attempt_test.go new file mode 100644 index 000000000..86feedce4 --- /dev/null +++ b/services/core/internal/execution/runtime_restore_attempt_test.go @@ -0,0 +1,238 @@ +package execution + +import ( + "context" + "encoding/json" + "errors" + "strings" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/google/uuid" +) + +type restoreAttemptProvider struct { + retentionProvider + outcome string + requests []sandbox.ResumeRequest + derived int + stop error +} + +func (p *restoreAttemptProvider) NewCompute(_ context.Context, _ sandbox.Reference, generation uint64, snapshot *sandbox.RetainedState) (sandbox.Compute, error) { + p.derived++ + return sandbox.Compute{Generation: generation, Name: "target", RestoredFrom: snapshot}, nil +} +func (p *restoreAttemptProvider) Resume(_ context.Context, q sandbox.ResumeRequest) (sandbox.ComputeState, error) { + p.requests = append(p.requests, q) + if !q.ReconcileOnly { + return sandbox.ComputeState{}, p.stop + } + if p.outcome == "unknown" { + return sandbox.ComputeState{}, sandbox.ErrComputeUnconfirmed + } + closed := q.OperationID + if p.outcome == "wrong_operation" { + closed = uuid.NewString() + } + result := sandbox.ComputeState{Compute: q.Target, Status: "absent", RestoreAttemptClosed: closed} + if p.outcome == "wrong_target" { + result.Compute.Name = "another-target" + } + return result, nil +} + +func TestRestoreAttemptRecoveryUsesExactClosedEvidence(t *testing.T) { + for _, outcome := range []string{"unsent_identity", "closed", "unknown", "wrong_operation", "wrong_target"} { + t.Run(outcome, func(t *testing.T) { + stop := errors.New("native request captured") + provider := &restoreAttemptProvider{outcome: outcome, stop: stop} + f := workspaceSettlementFixtureWithSetup(t, provider, workspaceNodeSetup(t, true)) + r, session := f.lifecycle, f.session + owner, err := r.provision(t.Context(), session.TenantID, session.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + snapshot := &sandbox.RetainedState{ID: "snapshot", Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}} + state := runtimeCompute{Version: sandbox.SuspensionStateVersion, Current: sandbox.Compute{ID: "source", Name: "source", Generation: 1}, Retained: snapshot, RestoreID: uuid.NewString()} + if outcome != "unsent_identity" { + state.Target = &sandbox.Compute{Name: "target", Generation: 2, RestoredFrom: snapshot} + } + raw, err := json.Marshal(state) + if err != nil { + t.Fatal(err) + } + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='restoring',compute_revision=4,compute_retained_until=clock_timestamp()+interval '1 hour',compute_state=$2 WHERE id=$1`, owner.ID, raw); err != nil { + t.Fatal(err) + } + owner, err = r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + err = r.restoreCompute(t.Context(), provider, owner, state, true) + current, readErr := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if readErr != nil { + t.Fatal(readErr) + } + var persisted runtimeCompute + if json.Unmarshal(current.ComputeState, &persisted) != nil { + t.Fatal("invalid persisted receipt") + } + switch outcome { + case "unsent_identity": + if !errors.Is(err, stop) || provider.derived != 1 || len(provider.requests) != 1 || provider.requests[0].ReconcileOnly || persisted.Target == nil || persisted.RestoreID != state.RestoreID { + t.Fatalf("unsent intent not safely derived: %v %#v", err, provider.requests) + } + case "closed": + if !errors.Is(err, stop) || len(provider.requests) != 2 || !provider.requests[0].ReconcileOnly || provider.requests[1].ReconcileOnly || persisted.RestoreID == state.RestoreID || provider.requests[1].OperationID != persisted.RestoreID { + t.Fatalf("closed attempt not replaced durably: %v %#v", err, provider.requests) + } + default: + want := sandbox.ErrOwnership + if outcome == "unknown" { + want = sandbox.ErrComputeUnconfirmed + } + if !errors.Is(err, want) || len(provider.requests) != 1 || persisted.RestoreID != state.RestoreID || current.ComputeRevision != owner.ComputeRevision { + t.Fatalf("unproven attempt was replayed: %v %#v", err, provider.requests) + } + } + }) + } +} + +func TestCheckpointTransferRoutesBeforeTargetIO(t *testing.T) { + for _, operation := range []string{"wake", "expiry"} { + t.Run(operation, func(t *testing.T) { + stop := errors.New("target Resume captured") + provider := &restoreAttemptProvider{stop: stop} + f := workspaceSettlementFixtureWithSetup(t, provider, workspaceNodeSetup(t, true)) + r, session := f.lifecycle, f.session + owner, err := r.provision(t.Context(), session.TenantID, session.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + view, err := r.deployments.View(t.Context()) + if err != nil { + t.Fatal(err) + } + token, err := r.deployments.CreateEnrollment(t.Context(), deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + if err != nil { + t.Fatal(err) + } + targetNode := uuid.NewString() + node := deployment.Enrollment{NodeID: targetNode, Name: "checkpoint target", Credential: strings.Repeat("t", 64), Provider: view.Provider, BackendFingerprint: strings.Repeat("c", 64), DeploymentGeneration: view.Generation, SpecificationDigest: view.SpecificationDigest, CoreURL: fixturePublicURL} + if _, err = r.deployments.Enroll(t.Context(), token.Token, node); err != nil { + t.Fatal(err) + } + connection := uuid.NewString() + if err = r.deployments.ConnectNode(t.Context(), targetNode, connection, view.OwnerEpoch); err != nil { + t.Fatal(err) + } + compat := sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"} + if err = r.deployments.Heartbeat(t.Context(), targetNode, connection, view.OwnerEpoch, deployment.NodeHealth{ProviderReady: true}, []sandbox.GenerationStatus{{Generation: view.Generation, SpecificationDigest: view.SpecificationDigest, State: "ready", Checkpoint: &compat}}); err != nil { + t.Fatal(err) + } + state := runtimeCompute{Version: sandbox.SuspensionStateVersion, Current: sandbox.Compute{ID: "source", Name: "source", Generation: 1}, Retained: &sandbox.RetainedState{ID: "snapshot", Compatibility: compat}} + raw, _ := json.Marshal(state) + if _, err = f.pool.Exec(t.Context(), `UPDATE environments SET initialization='complete',status='disconnected' WHERE id=$1`, owner.EnvironmentID); err != nil { + t.Fatal(err) + } + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspended',compute_state=$2,compute_revision=4,compute_retained_until=clock_timestamp()+interval '1 hour' WHERE id=$1`, owner.ID, raw); err != nil { + t.Fatal(err) + } + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET last_seen_at=clock_timestamp()-interval '1 minute' WHERE id=$1`, owner.NodeID); err != nil { + t.Fatal(err) + } + if operation == "wake" { + err = r.deployments.TouchActivity(t.Context(), owner.TenantID, owner.EnvironmentID) + } else { + _, err = f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID) + } + if err != nil { + t.Fatal(err) + } + source, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + hints := 0 + r.hintDestination = func(id string) { + hints++ + committed, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || id != targetNode || committed.NodeID != id || committed.ID != owner.ID { + t.Fatal("hint preceded durable route or targeted wrong node", committed, err) + } + } + if err = r.observe(t.Context(), source); err != nil { + t.Fatal(err) + } + moved, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + if moved.NodeID != targetNode || moved.ID != owner.ID || moved.DeviceID != owner.DeviceID || provider.derived != 0 || len(provider.requests) != 0 || provider.kills != 0 || provider.snapshots != 0 { + t.Fatal("source lane executed target effects or lost identity", moved) + } + if hints != 1 { + t.Fatalf("expected one committed handoff hint, got %d", hints) + } + // Simulate the independently scheduled target lane after the durable handoff. + r.nodeID = targetNode + err = r.observe(t.Context(), moved) + if operation == "wake" { + if !errors.Is(err, stop) || provider.derived != 1 || len(provider.requests) != 1 || provider.requests[0].ReconcileOnly { + t.Fatal("target failed to use committed unsent restore", err) + } + } else { + if err != nil { + t.Fatal(err) + } + released, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || released.State != "released" || provider.kills != 0 || provider.snapshots != 1 { + t.Fatal("offline source cleanup did not settle at archive domain", released, err) + } + } + }) + } +} + +// Archive bytes may be unusable while the exact durable ownership receipt is +// still valid for deletion. Observing it must not authorize a new capture. +type cleanupReceiptProvider struct { + retentionProvider + observed bool +} + +func (p *cleanupReceiptProvider) Suspend(_ context.Context, q sandbox.SuspendRequest) (sandbox.ComputeState, error) { + p.observed = q.ReconcileOnly + if !q.ReconcileOnly { + return sandbox.ComputeState{}, sandbox.ErrComputeUnconfirmed + } + return sandbox.ComputeState{Compute: q.Source, Status: "unknown", Retained: &sandbox.RetainedState{ID: "owned-corrupt-archive", OperationID: q.OperationID, Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}}}, nil +} +func TestLostCaptureReceiptCanStillBeDeleted(t *testing.T) { + p := &cleanupReceiptProvider{} + f := workspaceSettlementFixtureWithSetup(t, p, workspaceNodeSetup(t, true)) + r, s := f.lifecycle, f.session + owner, err := r.provision(t.Context(), s.TenantID, s.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + state := runtimeCompute{Version: sandbox.SuspensionStateVersion, Current: sandbox.Compute{ID: "source", Name: "source"}, SuspendID: uuid.NewString()} + raw, _ := json.Marshal(state) + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspending',compute_state=$2,compute_revision=4,compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID, raw); err != nil { + t.Fatal(err) + } + owner, err = r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + if err = r.observe(t.Context(), owner); err != nil { + t.Fatal(err) + } + current, err := r.reader.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || current.State != "released" || !p.observed || p.kills != 1 || p.snapshots != 1 { + t.Fatal("exact cleanup receipt was discarded", current, err, p.observed, p.kills, p.snapshots) + } +} diff --git a/services/core/internal/execution/runtime_retained_queries_test.go b/services/core/internal/execution/runtime_retained_queries_test.go new file mode 100644 index 000000000..e2430706b --- /dev/null +++ b/services/core/internal/execution/runtime_retained_queries_test.go @@ -0,0 +1,84 @@ +package execution + +import ( + "encoding/json" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/db/sqlc" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/google/uuid" + "github.com/jackc/pgx/v5/pgtype" +) + +// Exercise the SQL projections with the same serialized envelope Core persists. +// Adapter-private data deliberately contains legacy-looking keys: queries must +// read only the shared retained envelope, never the opaque payload. +func TestRuntimeRetainedStateFeedsCheckpointDemandAndNodeCounts(t *testing.T) { + f := newWorkspaceSettlementFixture(t, &retentionProvider{}) + r, session := f.lifecycle, f.session + owner, err := r.provision(t.Context(), session.TenantID, session.Environment.ID, r.config.InstallationID) + if err != nil { + t.Fatal(err) + } + compatibility := sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"} + source := sandbox.Compute{Generation: 1, Name: "source", ID: "native-source"} + retained := sandbox.RetainedState{Reference: "fixture/retained", ID: "native-retained", Data: `{"snapshot":{"Compatibility":{"artifact_domain":"private","execution_class":"private"}}}`, OperationID: "suspend-operation", SourceGeneration: source.Generation, SourceName: source.Name, SourceID: source.ID, Compatibility: compatibility} + for _, tc := range []struct { + name string + wake, expired, released, noCompatibility, legacy bool + demand, count int + }{ + {name: "waiting", wake: true, demand: 1, count: 1}, + {name: "idle", count: 1}, + {name: "expired", wake: true, expired: true, count: 1}, + {name: "released", wake: true, released: true}, + {name: "no_compatibility", wake: true, noCompatibility: true, count: 1}, + {name: "legacy_shape", wake: true, legacy: true}, + } { + t.Run(tc.name, func(t *testing.T) { + state := runtimeCompute{Version: sandbox.SuspensionStateVersion, Current: source, Retained: &retained, SuspendID: retained.OperationID} + copied := retained + if tc.noCompatibility { + copied.Compatibility = sandbox.CheckpointCompatibility{} + state.Retained = &copied + } + raw, err := json.Marshal(state) + if err != nil { + t.Fatal(err) + } + if tc.legacy { + var envelope map[string]json.RawMessage + if err := json.Unmarshal(raw, &envelope); err != nil { + t.Fatal(err) + } + envelope["snapshot"] = envelope["retained"] + delete(envelope, "retained") + raw, err = json.Marshal(envelope) + if err != nil { + t.Fatal(err) + } + } + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET state=CASE WHEN $3 THEN 'released' ELSE 'running' END,released_at=CASE WHEN $3 THEN clock_timestamp() END,compute_phase='suspended',compute_state=$2,compute_retained_until=clock_timestamp()+CASE WHEN $4 THEN interval '-1 hour' ELSE interval '1 hour' END,compute_wake_requested=$5 WHERE id=$1`, owner.ID, raw, tc.released, tc.expired, tc.wake); err != nil { + t.Fatal(err) + } + q := sqlc.New(f.pool) + demands, err := q.ListWaitingCheckpointRestores(t.Context()) + if err != nil || len(demands) != tc.demand { + t.Fatalf("checkpoint demand=%d want=%d: %v", len(demands), tc.demand, err) + } + if len(demands) == 1 { + var got sandbox.CheckpointCompatibility + if err := json.Unmarshal(demands[0].Checkpoint, &got); err != nil || got != compatibility { + t.Fatalf("shared compatibility lost: %+v %v", got, err) + } + } + nodes, err := q.ListRuntimeNodes(t.Context(), pgtype.UUID{Bytes: uuid.MustParse(owner.NodeID), Valid: true}) + if err != nil || len(nodes) != 1 { + t.Fatalf("node projection: %d %v", len(nodes), err) + } + if nodes[0].Snapshots != int64(tc.count) { + t.Fatalf("retained node count=%d want=%d", nodes[0].Snapshots, tc.count) + } + }) + } +} diff --git a/services/core/internal/execution/runtime_setup.go b/services/core/internal/execution/runtime_setup.go index ccfcb795f..1f1822177 100644 --- a/services/core/internal/execution/runtime_setup.go +++ b/services/core/internal/execution/runtime_setup.go @@ -3,6 +3,7 @@ package execution import ( "context" "errors" + "time" "github.com/MiniMax-AI/OpenAgentCore/internal/agentcapabilities" "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" @@ -85,7 +86,13 @@ func runRuntimeSetup(ctx context.Context, peer runtimePreparer, identity agentca if peer == nil { return errors.New("environment initialization request unavailable") } + deadline, bounded := ctx.Deadline() + budget := time.Until(deadline).Milliseconds() + if !bounded || budget <= 0 || budget > proto.RuntimePrepareMaxBudgetMS || ctx.Err() != nil { + return errors.New("environment initialization budget unavailable") + } request := operation.Request + request.BudgetMS = budget request.EnvironmentID, request.SessionID = identity.EnvironmentID, identity.SessionID result, err := peer.PrepareRuntime(ctx, uuid.NewString(), request, operation.Data) if err == nil && result.Outcome == "completed" { diff --git a/services/core/internal/execution/runtime_setup_test.go b/services/core/internal/execution/runtime_setup_test.go index be2a99461..2bd9be1bb 100644 --- a/services/core/internal/execution/runtime_setup_test.go +++ b/services/core/internal/execution/runtime_setup_test.go @@ -5,6 +5,7 @@ import ( "errors" "strings" "testing" + "time" "github.com/MiniMax-AI/OpenAgentCore/internal/agentcapabilities" "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" @@ -36,7 +37,7 @@ func TestRuntimeSetupReceiptOutcomes(t *testing.T) { } { t.Run(test.name, func(t *testing.T) { peer := &receiptRuntime{result: proto.RuntimePrepareResultPayload{Outcome: test.outcome, ExitCode: test.code}, err: test.err} - err := runRuntimeSetup(t.Context(), peer, agentcapabilities.Identity{}, runtimeSetupOperation{Request: proto.RuntimePreparePayload{Action: "initialize", Initialization: &proto.RuntimeInitialization{Action: "setup", Command: setupCanary}}}) + err := runRuntimeSetup(initializationTestContext(t), peer, agentcapabilities.Identity{}, runtimeSetupOperation{Request: proto.RuntimePreparePayload{Action: "initialize", Initialization: &proto.RuntimeInitialization{Action: "setup", Command: setupCanary}}}) if test.outcome == "completed" && test.err == nil { if err != nil { t.Fatal(err) @@ -73,14 +74,47 @@ func TestInitialFileUsesTypedRuntimeBytes(t *testing.T) { size := int64(len(body)) owner := agentcapabilities.Identity{EnvironmentID: "environment", SessionID: "session"} peer := &receiptRuntime{result: proto.RuntimePrepareResultPayload{Outcome: "completed"}} - if err := installInitialFile(t.Context(), peer, owner, environmentconfig.InitialFileMetadata{Path: "/workspace/a", SizeBytes: &size}, body); err != nil { + if err := installInitialFile(initializationTestContext(t), peer, owner, environmentconfig.InitialFileMetadata{Path: "/workspace/a", SizeBytes: &size}, body); err != nil { t.Fatal(err) } if peer.request.Action != "file" || peer.request.File.Path != "/workspace/a" || peer.request.EnvironmentID != owner.EnvironmentID || peer.request.SessionID != owner.SessionID || string(peer.data) != setupCanary { t.Fatal("file transport changed") } size++ - if err := installInitialFile(t.Context(), peer, owner, environmentconfig.InitialFileMetadata{SizeBytes: &size}, body); err == nil { + if err := installInitialFile(initializationTestContext(t), peer, owner, environmentconfig.InitialFileMetadata{SizeBytes: &size}, body); err == nil { t.Fatal("mismatched source size accepted") } } + +func initializationTestContext(t *testing.T) context.Context { + t.Helper() + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Minute) + t.Cleanup(cancel) + return ctx +} + +func TestRuntimeSetupUsesRemainingInitializationBudget(t *testing.T) { + ctx := initializationTestContext(t) + deadline, _ := ctx.Deadline() + peer := &receiptRuntime{result: proto.RuntimePrepareResultPayload{Outcome: "completed"}} + err := runRuntimeSetup(ctx, peer, agentcapabilities.Identity{}, runtimeSetupOperation{}) + if err != nil { + t.Fatal(err) + } + remaining := time.Until(deadline).Milliseconds() + if peer.request.BudgetMS < remaining || peer.request.BudgetMS > remaining+1000 || peer.request.BudgetMS <= 120000 { + t.Fatalf("remaining operation budget lost: %d vs %d", peer.request.BudgetMS, remaining) + } + for name, c := range map[string]context.Context{"unbounded": t.Context(), "expired": func() context.Context { + c, stop := context.WithDeadline(t.Context(), time.Now().Add(-time.Second)) + stop() + return c + }()} { + t.Run(name, func(t *testing.T) { + peer := &receiptRuntime{} + if runRuntimeSetup(c, peer, agentcapabilities.Identity{}, runtimeSetupOperation{}) == nil || peer.request.BudgetMS != 0 { + t.Fatal("invalid budget sent") + } + }) + } +} diff --git a/services/core/internal/execution/runtime_wake_hint_test.go b/services/core/internal/execution/runtime_wake_hint_test.go index 955afc33c..006871999 100644 --- a/services/core/internal/execution/runtime_wake_hint_test.go +++ b/services/core/internal/execution/runtime_wake_hint_test.go @@ -8,6 +8,8 @@ import ( "sync/atomic" "testing" "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" ) type maintenanceTestLoop struct { @@ -355,3 +357,30 @@ func TestRuntimeMaintenanceCancellationAfterSuccessfulScanWinsReadyTriggers(t *t t.Fatalf("ready triggers bypassed cancellation: calls=%d, error=%v", calls, err) } } + +func TestDestinationHintCreatesAndReusesOrdinaryLane(t *testing.T) { + m := &runtimeManager{ctx: t.Context(), config: RuntimeProvider{Mode: string(sandbox.DeploymentNodes), Provider: &retentionProvider{}}, nodes: map[string]*runtimeNode{}} + m.hintDestination("target") + n := m.nodes["target"] + if n == nil || n.lifecycle.hintDestination == nil || len(n.lifecycle.wakeHints) != 1 { + t.Fatal("missing destination lane or hint") + } + for range 10 { + m.hintDestination("target") + } + if m.nodes["target"] != n || len(n.lifecycle.wakeHints) != 1 { + t.Fatal("hint duplicated lane or failed to coalesce") + } + m.switching = true + m.hintDestination("during-transition") + if m.nodes["during-transition"] != nil { + t.Fatal("hint bypassed configuration transition") + } + m.switching = false + m.closed = true + m.hintDestination("after-close") + if m.nodes["after-close"] != nil { + t.Fatal("hint bypassed closed manager") + } + n.lifecycle.stop() +} diff --git a/services/core/internal/execution/runtime_workspace_settlement_test.go b/services/core/internal/execution/runtime_workspace_settlement_test.go index d4cf55108..5eef8ee81 100644 --- a/services/core/internal/execution/runtime_workspace_settlement_test.go +++ b/services/core/internal/execution/runtime_workspace_settlement_test.go @@ -21,6 +21,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/workspacefs" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/workspaces" "github.com/google/uuid" + "github.com/jackc/pgx/v5/pgxpool" ) // settledWorkspaceRefusal models the provider's pre-native filesystem rejection: @@ -42,7 +43,7 @@ type settlementWorkspaceControl struct { } func (c *settlementWorkspaceControl) Declaration() workspacefs.Declaration { - return workspacefs.Declaration{Attachment: workspacefs.AttachmentHostDirectory} + return workspacefs.Declaration{Attachment: workspacefs.AttachmentHostDirectory, UserXAttr: true} } func (c *settlementWorkspaceControl) Create(_ context.Context, ref workspacefs.Reference) (workspacefs.Attachment, error) { return workspacefs.Attachment{Reference: ref, ConfigurationID: c.configuration.ID, Kind: workspacefs.AttachmentHostDirectory, Native: []byte(`{}`)}, nil @@ -62,6 +63,7 @@ func (c settlementWorkspaceControls) Normalize(config workspacefs.Configuration) } type workspaceSettlementFixture struct { + pool *pgxpool.Pool lifecycle *runtimeLifecycle sessions *sessions.Service storage *workspacepg.Store @@ -70,6 +72,10 @@ type workspaceSettlementFixture struct { } func newWorkspaceSettlementFixture(t *testing.T, provider sandbox.SandboxProvider) workspaceSettlementFixture { + return workspaceSettlementFixtureWithSetup(t, provider, nil) +} + +func workspaceSettlementFixtureWithSetup(t *testing.T, provider sandbox.SandboxProvider, setup func(Owner, *deployment.Service, deployment.Reader) (string, string)) workspaceSettlementFixture { t.Helper() pool := pgtest.OpenIsolated(t, nil) @@ -80,7 +86,12 @@ func newWorkspaceSettlementFixture(t *testing.T, provider sandbox.SandboxProvide t.Cleanup(func() { _ = lease.Close(context.Background()) }) deployments, reader, operations := testDeployment(t, pool, pgtest.CredentialKey(t), lease) owner := Owner{Lease: lease, Deployment: operations, Sessions: sessionExecution(t, lease)} - installation := initializeE2BDeployment(t, owner) + var installation, nodeID string + if setup == nil { + installation, nodeID = workspaceNodeSetup(t, true)(owner, deployments, reader) + } else { + installation, nodeID = setup(owner, deployments, reader) + } projectID := uuid.NewString() audit := adminaudit.WithSource(t.Context(), adminaudit.Source{CredentialID: "fixture-admin", ProjectID: projectID, RequestID: uuid.NewString(), TraceID: uuid.NewString()}) management, err := projects.NewService(projectpg.New(pgunit.NewPool(pool))) @@ -92,11 +103,16 @@ func newWorkspaceSettlementFixture(t *testing.T, provider sandbox.SandboxProvide t.Fatal(err) } sessionReader, service := testSessions(t, pool, pgtest.CredentialKey(t)) - created, err := service.CreateSession(t.Context(), project.TenantID, sessions.CreateSession{Creator: identity.Subject{Kind: "service_account", ID: "fixture"}, Engine: "codex", IdempotencyKey: uuid.NewString(), Configuration: json.RawMessage(`{"agent":{"model":"test-model"},"environment":{"type":"openai_hosted","network":{"access":"disabled"}}}`), ModelProvider: &v1.ModelProviderInput{Protocol: "responses", BaseURL: "https://model.fixture.example/v1", APIKey: "fixture-key"}, ModelProviderSource: v1.ExecutionSourceSession}) + created, err := service.CreateSession(t.Context(), project.TenantID, sessions.CreateSession{SupportsRetainedNativeHistory: true, Creator: identity.Subject{Kind: "service_account", ID: "fixture"}, Engine: "codex", IdempotencyKey: uuid.NewString(), Configuration: json.RawMessage(`{"agent":{"model":"test-model"},"environment":{"type":"openai_hosted","network":{"access":"disabled"}}}`), ModelProvider: &v1.ModelProviderInput{Protocol: "responses", BaseURL: "https://model.fixture.example/v1", APIKey: "fixture-key"}, ModelProviderSource: v1.ExecutionSourceSession}) if err != nil { t.Fatal(err) } session := created.Session + if nodeID != "" { + if _, err := operations.EnsurePlacement(t.Context(), deployment.AllocationKey{TenantID: session.TenantID, EnvironmentID: session.Environment.ID}, installation); err != nil { + t.Fatal(err) + } + } configuration := workspacefs.Configuration{ID: uuid.NewString(), Adapter: "fixture", Parameters: []byte(`{}`)} storage := workspacepg.New(pgunit.NewPool(pool)) writer := workspacepg.NewExecution(lease) @@ -107,8 +123,8 @@ func newWorkspaceSettlementFixture(t *testing.T, provider sandbox.SandboxProvide control := &settlementWorkspaceControl{configuration: configuration} filesystems := workspaces.NewExecution(storage, writer, settlementWorkspaceControls{control}, lease) declaration := control.Declaration() - lifecycle := &runtimeLifecycle{registry: runtimegateway.NewRegistry(), sessions: sessionReader, sessionExecution: owner.Sessions, deployment: operations, deployments: deployments, reader: reader, lease: lease, workspaces: filesystems, config: RuntimeProvider{InstallationID: installation, Mode: "direct", Generation: 1, Provider: provider, Workspace: &declaration, WorkspaceRequirements: &workspacefs.Requirements{Attachment: workspacefs.AttachmentHostDirectory}}} - return workspaceSettlementFixture{lifecycle: lifecycle, sessions: service, storage: storage, control: control, session: session} + lifecycle := &runtimeLifecycle{nodeID: nodeID, registry: runtimegateway.NewRegistry(), sessions: sessionReader, sessionExecution: owner.Sessions, deployment: operations, deployments: deployments, reader: reader, lease: lease, workspaces: filesystems, config: RuntimeProvider{InstallationID: installation, Mode: "nodes", Generation: 1, Provider: provider, Workspace: &declaration, WorkspaceRequirements: &workspacefs.Requirements{Attachment: workspacefs.AttachmentHostDirectory}}} + return workspaceSettlementFixture{pool: pool, lifecycle: lifecycle, sessions: service, storage: storage, control: control, session: session} } func TestSettledWorkspaceRefusalRetainsReceiptUntilExplicitSessionDeletion(t *testing.T) { @@ -126,7 +142,7 @@ func TestSettledWorkspaceRefusalRetainsReceiptUntilExplicitSessionDeletion(t *te t.Fatal("missing durable failure", environment.Status, err) } replay, err := lifecycle.provision(t.Context(), session.TenantID, session.Environment.ID, installation) - if err != nil || !replay.Replayed || replay.ID != allocation.ID || provider.creates != 1 { + if !errors.Is(err, deployment.ErrAllocationConflict) || replay.ID != "" || provider.creates != 1 { t.Fatal("settled failed creation retried compute", replay, err) } if _, err = filesystems.DeleteBatch(t.Context(), ""); err != nil || control.deletes != 0 { diff --git a/services/core/internal/execution/runtime_workspace_test.go b/services/core/internal/execution/runtime_workspace_test.go index 0999b7c99..70210a6f6 100644 --- a/services/core/internal/execution/runtime_workspace_test.go +++ b/services/core/internal/execution/runtime_workspace_test.go @@ -93,7 +93,9 @@ func TestRuntimeWorkspaceConvergesBeforeComputeReservation(t *testing.T) { } return stop }}) - r := &runtimeLifecycle{sessions: workspaceEnvironmentReader{}, deployment: operations, reader: &strictDeploymentReader{t: t, environmentAllocation: func(context.Context, deployment.AllocationKey) (deployment.Allocation, error) { + r := &runtimeLifecycle{sessions: workspaceEnvironmentReader{}, deployment: operations, reader: &strictDeploymentReader{t: t, lifecyclePlacement: func(context.Context, deployment.AllocationKey) (deployment.LifecyclePlacement, error) { + return deployment.LifecyclePlacement{Specification: []byte(`{"workspace":{"attachment":"host_directory"}}`)}, nil + }, environmentAllocation: func(context.Context, deployment.AllocationKey) (deployment.Allocation, error) { return deployment.Allocation{}, deployment.ErrNotFound }}, workspaces: workspaces.NewExecution(f, f, workspaceControls{f}, heldLease{}), config: RuntimeProvider{Mode: "direct", Generation: 1, InstallationID: uuid.NewString(), Workspace: &workspacefs.Declaration{Attachment: workspacefs.AttachmentHostDirectory}, WorkspaceRequirements: &workspacefs.Requirements{Attachment: workspacefs.AttachmentHostDirectory}}} _, err := r.provision(t.Context(), f.record.Reference.TenantID, f.record.Reference.EnvironmentID, r.config.InstallationID) @@ -151,7 +153,9 @@ func TestOwnedGenerationDoesNotAdoptLaterFilesystemSelection(t *testing.T) { _, operations := deploymentOperations(t, &strictDeploymentStorage{t: t}, &strictDeploymentReader{t: t}, &strictExecutionStorage{t: t, withReservation: func(context.Context, deployment.AllocationKey, func(sessions.LockedSession, deployment.ReservationTx) error) error { return stop }}) - r := &runtimeLifecycle{sessions: workspaceEnvironmentReader{}, deployment: operations, reader: &strictDeploymentReader{t: t, environmentAllocation: func(context.Context, deployment.AllocationKey) (deployment.Allocation, error) { + r := &runtimeLifecycle{sessions: workspaceEnvironmentReader{}, deployment: operations, reader: &strictDeploymentReader{t: t, lifecyclePlacement: func(context.Context, deployment.AllocationKey) (deployment.LifecyclePlacement, error) { + return deployment.LifecyclePlacement{Specification: []byte(`{}`)}, nil + }, environmentAllocation: func(context.Context, deployment.AllocationKey) (deployment.Allocation, error) { return deployment.Allocation{}, deployment.ErrNotFound }}, workspaces: workspaces.NewExecution(f, f, workspaceControls{f}, heldLease{}), config: RuntimeProvider{Mode: "direct", Generation: 1, InstallationID: uuid.NewString()}} _, err := r.provision(t.Context(), f.record.Reference.TenantID, f.record.Reference.EnvironmentID, r.config.InstallationID) diff --git a/services/core/internal/execution/support.go b/services/core/internal/execution/support.go index 388a4d544..0a264fd7e 100644 --- a/services/core/internal/execution/support.go +++ b/services/core/internal/execution/support.go @@ -1,6 +1,7 @@ package execution import ( + "context" "encoding/json" "errors" "strings" @@ -144,3 +145,32 @@ func (p Policy) engineCapabilities(peer *runtimegateway.Session, engine string, } return caps, nil } + +// sessionCapabilities shares storage-specific validation between device selection +// and the final preparation gate. The immutable binding, not today's deployment, +// determines whether this Session requires native history outside compute. +func (d *Dispatcher) sessionCapabilities(ctx context.Context, peer *runtimegateway.Session, session sessions.Session, snapshot Snapshot) (runtimedevice.KindCapabilities, error) { + caps, err := d.engineCapabilities(peer, session.Engine, snapshot) + if err != nil { + return caps, err + } + if snapshot.Environment == nil || snapshot.Environment.Type != "openai_hosted" { + return caps, nil + } + environment, err := d.SessionsReader.GetSessionEnvironment(ctx, session.TenantID, session.ID) + if err != nil { + return runtimedevice.KindCapabilities{}, err + } + if err := d.validateEnvironmentHistory(session.Engine, environment, caps.RetainedNativeHistory); err != nil { + return runtimedevice.KindCapabilities{}, err + } + return caps, nil +} + +func (p Policy) validateEnvironmentHistory(engine string, environment sessions.Environment, supported bool) error { + profile, found := p.Engines.Lookup(engine) + if !found { + return sessions.ErrInvalidInput + } + return sessions.ValidateRetainedHistory(environment.ExternalWorkspace, supported && profile.RetainedNativeHistory.IsSupported()) +} diff --git a/services/core/internal/execution/tool_search_test.go b/services/core/internal/execution/tool_search_test.go index 80a5ea7dc..314098b0f 100644 --- a/services/core/internal/execution/tool_search_test.go +++ b/services/core/internal/execution/tool_search_test.go @@ -22,6 +22,7 @@ func TestDiscoveryUsesSharedOperationQualification(t *testing.T) { workspace := strings.Replace(discoveryConfiguration, `"type":"none"`, `"type":"openai_hosted"`, 1) policy := Policy{Engines: engine.NewCatalog(map[string]engine.Profile{"new_harness": enginetest.Profile(func(p *engine.Profile) { p.Placements = []string{"openai_hosted"} + p.RetainedNativeHistory = proto.CapabilitySupported p.ToolSearch = proto.CapabilitySupported })})} if err := policy.ValidateSessionConfiguration("new_harness", json.RawMessage(workspace)); err != nil { diff --git a/services/core/internal/execution/worker.go b/services/core/internal/execution/worker.go index 919740614..6419f31af 100644 --- a/services/core/internal/execution/worker.go +++ b/services/core/internal/execution/worker.go @@ -155,12 +155,17 @@ func (w *Worker) CreateSession(ctx context.Context, tenant string, input session if err := w.validateCreation(ctx, input); err != nil { return sessions.Creation{}, err } + profile, ok := w.dispatcher.Engines.Lookup(input.Engine) + if !ok { + return sessions.Creation{}, sessions.ErrInvalidInput + } + input.SupportsRetainedNativeHistory = profile.RetainedNativeHistory.IsSupported() creation, err := w.dispatcher.Sessions.CreateSession(ctx, tenant, input) if err == nil { w.hintRuntimeWake(ctx, creation.Session) - } - if err == nil && len(input.InitialInputs) > 0 { - recordInitialInputOrigin(ctx, creation.Session.ID) + if len(input.InitialInputs) > 0 { + recordInitialInputOrigin(ctx, creation.Session.ID) + } w.wakeScheduler() } return creation, err diff --git a/services/core/internal/execution/worker_device.go b/services/core/internal/execution/worker_device.go index 015ac962e..dcc634ddf 100644 --- a/services/core/internal/execution/worker_device.go +++ b/services/core/internal/execution/worker_device.go @@ -58,7 +58,7 @@ func (w *Worker) bindDevice(ctx context.Context, tenantID, sessionID string, inp return false, err } return w.bindSessionDevice(ctx, session, func(id string) bool { - if !w.ready(ctx, id, session.Engine, snapshot) { + if !w.ready(ctx, id, session, snapshot) { return false } if !input.HasImages() { @@ -121,11 +121,11 @@ func (w *Worker) bindSessionDevice(ctx context.Context, session sessions.Session return false, nil } -func (w *Worker) ready(ctx context.Context, deviceID, engine string, snapshot Snapshot) bool { +func (w *Worker) ready(ctx context.Context, deviceID string, session sessions.Session, snapshot Snapshot) bool { peer, err := w.dispatcher.authorizedPeer(ctx, deviceID) if err != nil { return false } - _, err = w.dispatcher.engineCapabilities(peer, engine, snapshot) + _, err = w.dispatcher.sessionCapabilities(ctx, peer, session, snapshot) return err == nil } diff --git a/services/core/internal/execution/worker_wakeup.go b/services/core/internal/execution/worker_wakeup.go index 153a6b962..c210235d2 100644 --- a/services/core/internal/execution/worker_wakeup.go +++ b/services/core/internal/execution/worker_wakeup.go @@ -9,6 +9,12 @@ import ( // wakeScheduler only hints at committed work. The existing loop retains lease, // capacity and Session ownership; polling recovers absent or coalesced hints. func (w *Worker) wakeScheduler() { + if w.runtimes != nil { + select { + case w.runtimes.placementWake <- struct{}{}: + default: + } + } select { case w.scheduleWake <- struct{}{}: default: diff --git a/services/core/internal/execution/worker_wakeup_test.go b/services/core/internal/execution/worker_wakeup_test.go index 92d6ed910..b56804627 100644 --- a/services/core/internal/execution/worker_wakeup_test.go +++ b/services/core/internal/execution/worker_wakeup_test.go @@ -8,7 +8,7 @@ import ( ) func TestSchedulerWakeCoalescesConcurrentAdmissionsAndKeepsNextHint(t *testing.T) { - worker := &Worker{scheduleWake: make(chan struct{}, 1)} + worker := &Worker{scheduleWake: make(chan struct{}, 1), runtimes: &runtimeManager{placementWake: make(chan struct{}, 1)}} var callers sync.WaitGroup for range 100 { callers.Go(func() { @@ -21,10 +21,19 @@ func TestSchedulerWakeCoalescesConcurrentAdmissionsAndKeepsNextHint(t *testing.T if got := len(worker.scheduleWake); got != 1 { t.Fatalf("queued wakeups = %d, want one", got) } + if got := len(worker.runtimes.placementWake); got != 1 { + t.Fatalf("placement wakeups = %d, want one", got) + } + <-worker.runtimes.placementWake <-worker.scheduleWake // A commit while a previous scan is running needs a subsequent scan. worker.wakeScheduler() select { + case <-worker.runtimes.placementWake: + default: + t.Fatal("admission during placement scan lost its wakeup") + } + select { case <-worker.scheduleWake: default: t.Fatal("admission during a scan lost its wakeup") diff --git a/services/core/internal/persistence/postgres/deploymentpg/allocations.go b/services/core/internal/persistence/postgres/deploymentpg/allocations.go index b65e310ee..bcbd6ab4c 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/allocations.go +++ b/services/core/internal/persistence/postgres/deploymentpg/allocations.go @@ -34,7 +34,7 @@ func allocation(row sqlc.RuntimeAllocation, session, tenant pgtype.UUID, deleted // findAllocation reads the tenant's allocation of the Environment; found is // false when it has none. func findAllocation(ctx context.Context, q *sqlc.Queries, tenant, environment pgtype.UUID) (deployment.Allocation, bool, error) { - row, err := q.GetRuntimeAllocation(ctx, sqlc.GetRuntimeAllocationParams{TenantID: tenant, EnvironmentID: environment}) + row, err := q.GetLatestRuntimeAllocation(ctx, sqlc.GetLatestRuntimeAllocationParams{TenantID: tenant, EnvironmentID: environment}) if errors.Is(err, pgx.ErrNoRows) { return deployment.Allocation{}, false, nil } @@ -117,7 +117,7 @@ func (e *Execution) withAllocation(ctx context.Context, key deployment.Allocatio } return translate(e.lease.Transaction(ctx, func(ctx context.Context, tx pgx.Tx) error { q := sqlc.New(tx) - row, err := q.GetRuntimeAllocation(ctx, sqlc.GetRuntimeAllocationParams{TenantID: tenant, EnvironmentID: environment}) + row, err := q.GetLatestRuntimeAllocation(ctx, sqlc.GetLatestRuntimeAllocationParams{TenantID: tenant, EnvironmentID: environment}) if errors.Is(err, pgx.ErrNoRows) { return deployment.ErrNotFound } @@ -164,7 +164,37 @@ func (s *Store) WithActivity(ctx context.Context, key deployment.AllocationKey, return err } touch := activityTx(func() error { - return q.TouchRuntimeActivity(ctx, sqlc.TouchRuntimeActivityParams{TenantID: tenant, EnvironmentID: environment}) + retained := pgtype.UUID{} + current, found, err := findAllocation(ctx, q, tenant, environment) + if err != nil { + return err + } + if found && current.State == "released" { + d, err := placementpg.LockDeployment(ctx, q) + if err != nil { + return err + } + if d.Resetting { + return placement.ErrResetAdmission + } + id, err := parseID(current.ID) + if err != nil { + return err + } + replaceable, err := q.CanReplaceRuntimeAllocation(ctx, id) + if err != nil { + return err + } + qualified, err := q.CanRetainRuntimeEnvironment(ctx, id) + if err != nil { + return err + } + if !replaceable || !qualified { + return deployment.ErrAllocationConflict + } + retained = id + } + return q.TouchRuntimeActivity(ctx, sqlc.TouchRuntimeActivityParams{TenantID: tenant, EnvironmentID: environment, RetainedID: retained}) }) if err := apply(locked, touch); err != nil { return err @@ -195,7 +225,52 @@ func (t *reservationTx) LockDeployment() (placement.Deployment, error) { } func (t *reservationTx) LoadReserved() (placement.Reserved, error) { - return placementpg.LoadReserved(t.ctx, t.q, t.environment) + reserved, err := placementpg.LoadReserved(t.ctx, t.q, t.environment) + if errors.Is(err, pgx.ErrNoRows) { + return placement.Reserved{Released: true}, nil + } + return reserved, err +} + +func (t *reservationTx) CanReplaceAllocation() (bool, error) { + current, found, err := t.FindAllocation() + if err != nil || !found { + return false, err + } + id, err := parseID(current.ID) + if err != nil { + return false, err + } + allowed, err := t.q.CanReplaceRuntimeAllocation(t.ctx, id) + if err != nil || !allowed { + return false, err + } + return t.q.CanRetainRuntimeEnvironment(t.ctx, id) +} + +func (t *reservationTx) LoadNodes() ([]placement.Node, error) { + return placementpg.LoadNodes(t.ctx, t.q) +} + +func (t *reservationTx) ReservePlacement(p placement.Placement) error { + _, found, err := t.FindAllocation() + if err != nil { + return err + } + if !found { + return placementpg.ReservePlacement(t.ctx, t.q, t.session, p) + } + node, err := parseID(p.NodeID) + if err != nil { + return err + } + count, err := t.q.ReserveReleasedRuntimePlacement(t.ctx, sqlc.ReserveReleasedRuntimePlacementParams{ + EnvironmentID: t.environment, NodeID: node, DeploymentGeneration: pgtype.Int8{Int64: int64(p.Generation), Valid: true}, + }) + if err == nil && count != 1 { + return deployment.ErrAllocationConflict + } + return err } func (t *reservationTx) InsertAllocation(a deployment.NewAllocation) (deployment.Allocation, error) { @@ -261,12 +336,31 @@ func (t *allocationTx) LoadActivity(current deployment.Allocation) (deployment.A return loadActivity(t.ctx, t.q, id) } -func (t *allocationTx) LoadRestore(current deployment.Allocation) (placement.Restore, error) { - node, err := parseID(current.NodeID) +func (t *allocationTx) LoadCheckpointPlacement(current deployment.Allocation) (placement.Deployment, []placement.CheckpointNode, error) { + d, err := placementpg.LockDeployment(t.ctx, t.q) if err != nil { - return placement.Restore{}, err + return d, nil, err } - return placementpg.LoadRestore(t.ctx, t.q, node, current.DeploymentGeneration) + nodes, err := placementpg.LoadNodes(t.ctx, t.q) + if err != nil { + return d, nil, err + } + candidates, err := placementpg.LoadCheckpointNodes(t.ctx, t.q, current.DeploymentGeneration, nodes) + return d, candidates, err +} + +func (t *allocationTx) MoveSuspended(current deployment.Allocation, nodeID string, change deployment.ComputeChange) (deployment.Allocation, error) { + source, err := parseID(current.NodeID) + if err != nil { + return deployment.Allocation{}, err + } + destination, err := parseID(nodeID) + if err != nil { + return deployment.Allocation{}, err + } + return t.change(current, func(ctx context.Context, id pgtype.UUID) (sqlc.RuntimeAllocation, error) { + return t.q.MoveSuspendedRuntimeCompute(ctx, sqlc.MoveSuspendedRuntimeComputeParams{ID: id, Source: source, Destination: destination, Revision: current.ComputeRevision, Phase: change.Phase, State: change.State}) + }) } func (t *allocationTx) ObserveRunning(current deployment.Allocation) (deployment.Allocation, error) { @@ -333,6 +427,14 @@ type cleanupTx struct { *sessionpg.SessionTx } +func (t *allocationTx) CanRetainEnvironment(current deployment.Allocation) (bool, error) { + id, err := parseID(current.ID) + if err != nil { + return false, err + } + return t.q.CanRetainRuntimeEnvironment(t.ctx, id) +} + func (t *cleanupTx) RevokeDevice(current deployment.Allocation) error { device, err := parseID(current.DeviceID) if err != nil { @@ -513,6 +615,24 @@ func (s *Store) UnallocatedEnvironments(ctx context.Context, nodeID, after strin return result, nil } +func (s *Store) RetainedNativeHistory(ctx context.Context, key deployment.AllocationKey) (bool, error) { + current, err := s.EnvironmentAllocation(ctx, key) + if err != nil { + return false, err + } + id, err := parseID(current.ID) + if err != nil { + return false, err + } + return s.pool.Queries().CanRetainRuntimeEnvironment(ctx, id) +} + +func (s *Store) PlacementDemand(ctx context.Context, after deployment.PlacementDemandCursor) ([]deployment.PlacementDemand, deployment.PlacementDemandCursor, error) { + ctx, cancel := context.WithTimeout(ctx, pgunit.ExecutionTimeout) + defer cancel() + return loadPlacementDemand(ctx, s.pool.Queries(), after) +} + func (s *Store) LifecyclePlacement(ctx context.Context, key deployment.AllocationKey) (deployment.LifecyclePlacement, error) { tenant, environment, err := allocationKey(key) if err != nil { @@ -527,7 +647,7 @@ func (s *Store) LifecyclePlacement(ctx context.Context, key deployment.Allocatio if err != nil { return deployment.LifecyclePlacement{}, err } - return deployment.LifecyclePlacement{Provider: row.ProviderKind, Mode: row.Mode, AllocationID: uuidString(row.AllocationID), AllocationNodeID: uuidString(row.AllocationNodeID), + return deployment.LifecyclePlacement{Specification: row.Specification, Provider: row.ProviderKind, Mode: row.Mode, AllocationID: uuidString(row.AllocationID), AllocationNodeID: uuidString(row.AllocationNodeID), PlacementNodeID: uuidString(row.PlacementNodeID), PlacementReleased: row.ReleasedAt.Valid}, nil } @@ -554,3 +674,7 @@ func (s *Store) CountRetainedAllocations(ctx context.Context, installationID str } return s.pool.Queries().CountRuntimeRetainedAllocations(ctx, id) } + +func (t *reservationTx) LoadGenerationSpecification(generation uint64) (deployment.GenerationSpecification, error) { + return (unit{ctx: t.ctx, q: t.q}).LoadGenerationSpecification(generation) +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/checkpoint_transfer_test.go b/services/core/internal/persistence/postgres/deploymentpg/checkpoint_transfer_test.go new file mode 100644 index 000000000..63f150c89 --- /dev/null +++ b/services/core/internal/persistence/postgres/deploymentpg/checkpoint_transfer_test.go @@ -0,0 +1,301 @@ +package deploymentpg_test + +import ( + "encoding/json" + "errors" + "github.com/google/uuid" + "sync" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" +) + +var checkpointClass = sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-archive", ExecutionClass: "fixture-machine"} + +func readyCheckpointNode(t *testing.T, f fixture, node deployment.Enrollment, view deployment.View, compatibility sandbox.CheckpointCompatibility) { + t.Helper() + connection := uuid.NewString() + epoch, err := f.adapter.OwnerEpoch(t.Context()) + if err != nil { + t.Fatal(err) + } + if err := f.service.ConnectNode(t.Context(), node.NodeID, connection, epoch); err != nil { + t.Fatal(err) + } + if err := f.service.HeartbeatGenerations(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{}, []sandbox.GenerationStatus{{Generation: view.Generation, SpecificationDigest: view.SpecificationDigest, State: "ready", Checkpoint: &compatibility}}); err != nil { + t.Fatal(err) + } +} + +func suspendedCheckpointOwner(t *testing.T, f fixture, changes *deployment.ExecutionOperations, installation string, node deployment.Enrollment, view deployment.View) deployment.Allocation { + t.Helper() + owner := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspended',compute_state='{"protocol_version":"1","retained":{"ID":"snapshot","Compatibility":{"artifact_domain":"fixture-archive","execution_class":"fixture-machine"}}}',compute_retained_until=clock_timestamp()+interval '24 hours',compute_wake_requested=true WHERE id=$1`, owner.ID); err != nil { + t.Fatal(err) + } + owner, err := f.adapter.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + return owner +} + +func TestCheckpointTransferPreservesAllocationAndFencesOldOwner(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, source, view, checkpointClass) + target := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, target, view, checkpointClass) + owner := suspendedCheckpointOwner(t, f, changes, installation, source, view) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + intent := json.RawMessage(`{"snapshot":{"ID":"snapshot"},"restore_id":"attempt"}`) + moved, err := changes.BeginRestore(t.Context(), owner, checkpointClass, intent) + if err != nil { + t.Fatal(err) + } + if moved.ID != owner.ID || moved.DeviceID != owner.DeviceID || moved.DeploymentGeneration != owner.DeploymentGeneration || moved.NodeID != target.NodeID || moved.ComputePhase != "restoring" || moved.ComputeRevision != owner.ComputeRevision+1 || !moved.ComputeRetainedUntil.Equal(*owner.ComputeRetainedUntil) { + t.Fatal("lost restore ownership", moved) + } + var placementNode, credentialNode string + if err := f.pool.QueryRow(t.Context(), `SELECT p.node_id::text,a.node_id::text FROM runtime_placements p JOIN runtime_allocations a ON a.environment_id=p.environment_id JOIN devices d ON d.id=a.device_id WHERE a.id=$1 AND d.revoked_at IS NULL`, owner.ID).Scan(&placementNode, &credentialNode); err != nil || placementNode != target.NodeID || credentialNode != target.NodeID { + t.Fatal("placement/device routing split", placementNode, credentialNode, err) + } + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil { + t.Fatal(err) + } + for _, n := range nodes { + if n.ID == source.NodeID && (n.Active != 0 || n.Retained != 0) || n.ID == target.NodeID && (n.Active != 1 || n.Retained != 1) { + t.Fatal("incorrect capacity", n) + } + } + if _, err := changes.BeginRestore(t.Context(), owner, checkpointClass, intent); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("stale source moved again", err) + } + if _, err := changes.BeginRestore(t.Context(), moved, checkpointClass, intent); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("unknown restore moved again", err) + } + if _, err := changes.RelocateCleanup(t.Context(), moved, checkpointClass); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("unknown restore reassigned cleanup", err) + } + +} + +func TestCheckpointTransferRequiresCompatibleGenerationAndCapacity(t *testing.T) { + for _, obstruction := range []string{"archive", "execution", "generation", "connection", "offline", "retained full", "active full", "expired", "quiescing", "no demand", "reset", "deleted", "epoch"} { + t.Run(obstruction, func(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + readyCheckpointNode(t, f, source, view, checkpointClass) + target := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, target, view, checkpointClass) + owner := suspendedCheckpointOwner(t, f, changes, installation, source, view) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + var statement string + arg := target.NodeID + want := placement.ErrNodeUnavailable + switch obstruction { + case "archive": + statement = `UPDATE runtime_node_generation_status SET checkpoint='{"artifact_domain":"other","execution_class":"fixture-machine"}' WHERE node_id=$1` + case "execution": + statement = `UPDATE runtime_node_generation_status SET checkpoint='{"artifact_domain":"fixture-archive","execution_class":"other"}' WHERE node_id=$1` + case "generation": + statement = `UPDATE runtime_node_generation_status SET generation=generation+1 WHERE node_id=$1` + case "connection": + statement = `UPDATE runtime_node_generation_status SET connection_id=gen_random_uuid() WHERE node_id=$1` + case "offline": + statement = `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1` + case "retained full": + suspendedCheckpointOwner(t, f, changes, installation, target, view) + case "active full": + runningPressureOwner(t, f, changes, installation, target.NodeID, view.Generation) + case "expired": + statement = `UPDATE runtime_allocations SET compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1` + arg = owner.ID + want = deployment.ErrAllocationConflict + case "quiescing": + statement = `UPDATE runtime_allocations SET compute_phase='quiescing' WHERE id=$1` + arg = owner.ID + want = deployment.ErrAllocationConflict + case "no demand": + statement = `UPDATE runtime_allocations SET compute_wake_requested=false WHERE id=$1` + arg = owner.ID + want = deployment.ErrAllocationConflict + case "deleted": + statement = `UPDATE sessions SET deleted_at=clock_timestamp() WHERE id=(SELECT session_id FROM environments WHERE id=$1)` + arg = owner.EnvironmentID + want = sessions.ErrNotFound + case "epoch": + statement = `UPDATE runtime_node_generation_status SET owner_epoch=owner_epoch+1 WHERE node_id=$1` + case "reset": + statement = `UPDATE runtime_deployment SET reset_clear='force',reset_requested_at=clock_timestamp(),reset_forced_at=clock_timestamp(),reset_audit='{}' WHERE installation_id=$1` + arg = installation + want = placement.ErrResetAdmission + } + if statement != "" { + if _, err := f.pool.Exec(t.Context(), statement, arg); err != nil { + t.Fatal(err) + } + } + if _, err := changes.BeginRestore(t.Context(), owner, checkpointClass, json.RawMessage(`{"restore_id":"attempt"}`)); !errors.Is(err, want) { + t.Fatal("invalid destination admitted", err, want) + } + current, err := f.adapter.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || current.NodeID != source.NodeID || current.ComputeRevision != owner.ComputeRevision { + t.Fatal("rejection changed ownership", current, err) + } + }) + } +} + +func TestCheckpointTransfersSerializeDestinationCapacity(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 2, MaxRetained: 2}) + readyCheckpointNode(t, f, source, view, checkpointClass) + target := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, target, view, checkpointClass) + owners := []deployment.Allocation{suspendedCheckpointOwner(t, f, changes, installation, source, view), suspendedCheckpointOwner(t, f, changes, installation, source, view)} + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + errs := make([]error, len(owners)) + var wg sync.WaitGroup + for i, owner := range owners { + wg.Go(func() { + _, errs[i] = changes.BeginRestore(t.Context(), owner, checkpointClass, json.RawMessage(`{"restore_id":"attempt"}`)) + }) + } + wg.Wait() + success := 0 + for _, err := range errs { + if err == nil { + success++ + } else if !errors.Is(err, placement.ErrNodeUnavailable) { + t.Fatal(err) + } + } + if success != 1 { + t.Fatal("destination overbooked", errs) + } +} + +func TestCheckpointCleanupRelocatesArtifactWithoutExecutionCompatibility(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, source, view, checkpointClass) + target := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + other := checkpointClass + other.ExecutionClass = "other-machine" + readyCheckpointNode(t, f, target, view, other) + runningPressureOwner(t, f, changes, installation, target.NodeID, view.Generation) + owner := suspendedCheckpointOwner(t, f, changes, installation, source, view) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + if _, err := changes.RelocateCleanup(t.Context(), owner, checkpointClass); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("moved live allocation for cleanup", err) + } + cleanup, err := changes.RequestCleanup(t.Context(), owner) + if err != nil { + t.Fatal(err) + } + moved, err := changes.RelocateCleanup(t.Context(), cleanup, checkpointClass) + if err != nil { + t.Fatal(err) + } + if moved.NodeID != target.NodeID || moved.State != "cleanup_pending" || moved.ComputePhase != "suspended" || moved.ID != owner.ID || !moved.ComputeRetainedUntil.Equal(*owner.ComputeRetainedUntil) { + t.Fatal("cleanup recreated compute", moved) + } + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil { + t.Fatal(err) + } + for _, n := range nodes { + if n.ID == target.NodeID && (n.Active != 1 || n.Retained != 2) { + t.Fatal("cleanup released retained slot", n) + } + } + if _, err := changes.ReleaseAllocation(t.Context(), moved); err != nil { + t.Fatal(err) + } +} + +func TestCheckpointPressureSuspendsOnlyUsableDestination(t *testing.T) { + for _, scenario := range []string{"waiting restore", "retained full", "incompatible", "other free"} { + t.Run(scenario, func(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, source, view, checkpointClass) + suspendedCheckpointOwner(t, f, changes, installation, source, view) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + retained := 2 + if scenario == "retained full" { + retained = 1 + } + target := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: retained}) + compatibility := checkpointClass + if scenario == "incompatible" { + compatibility.ExecutionClass = "other" + } + readyCheckpointNode(t, f, target, view, compatibility) + active := runningPressureOwner(t, f, changes, installation, target.NodeID, view.Generation) + if scenario == "other free" { + other := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, other, view, checkpointClass) + } + until := time.Now().Add(24 * time.Hour) + _, err := changes.SetCompute(t.Context(), active, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute) + if scenario == "waiting restore" { + if err != nil { + t.Fatal("compatible wake cannot trigger pressure suspension", err) + } + } else if !errors.Is(err, deployment.ErrNotIdle) { + t.Fatal("suspended without usable restore capacity", err) + } + }) + } +} + +func TestCheckpointRestoreUsesRetainedSlotAndImmutableGeneration(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + source := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + readyCheckpointNode(t, f, source, view, checkpointClass) + owner := suspendedCheckpointOwner(t, f, changes, installation, source, view) + // A rollout must not substitute the serving generation for this snapshot. + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET ready_generation=ready_generation+1 WHERE id=$1`, source.NodeID); err != nil { + t.Fatal(err) + } + restored, err := changes.BeginRestore(t.Context(), owner, checkpointClass, json.RawMessage(`{"restore_id":"attempt"}`)) + if err != nil { + t.Fatal(err) + } + if restored.NodeID != source.NodeID || restored.DeploymentGeneration != owner.DeploymentGeneration { + t.Fatal("changed retained generation", restored) + } + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil || len(nodes) != 1 || nodes[0].Active != 1 || nodes[0].Retained != 1 { + t.Fatal("restore reserved a second retained slot", nodes, err) + } +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/fixture_test.go b/services/core/internal/persistence/postgres/deploymentpg/fixture_test.go index 45c48d97a..8032f8288 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/fixture_test.go +++ b/services/core/internal/persistence/postgres/deploymentpg/fixture_test.go @@ -6,18 +6,18 @@ import ( "strings" "testing" - "github.com/google/uuid" - "github.com/jackc/pgx/v5/pgxpool" - "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/adminaudit" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/credentialcrypto" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/deploymentpg" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/pgtest" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/pgunit" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/providers" + "github.com/google/uuid" + "github.com/jackc/pgx/v5/pgxpool" ) const fixturePublicURL = "https://core.example" @@ -77,7 +77,7 @@ func (f fixture) execution(t *testing.T) (*deployment.ExecutionOperations, *pgun t.Fatal(err) } t.Cleanup(func() { _ = lease.Close(context.Background()) }) - operations, err := deployment.NewExecutionOperations(f.service, deploymentpg.NewExecution(lease, f.cipher)) + operations, err := deployment.NewExecutionOperations(f.service, deploymentpg.NewExecution(lease, f.cipher), engine.Catalog{}) if err != nil { t.Fatal(err) } @@ -145,8 +145,22 @@ func (f fixture) connect(t *testing.T, nodeID string) string { if err := f.service.ConnectNode(t.Context(), nodeID, connection, epoch); err != nil { t.Fatal(err) } - if err := f.service.Heartbeat(t.Context(), nodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := f.heartbeat(t.Context(), nodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}, nil); err != nil { t.Fatal(err) } return connection } + +// heartbeat reports the fixture's verified generation alongside static readiness. +func (f fixture) heartbeat(ctx context.Context, nodeID, connection string, epoch uint64, health deployment.NodeHealth, _ []sandbox.GenerationStatus) error { + var generation uint64 + var digest, provider string + if err := f.pool.QueryRow(ctx, "SELECT n.deployment_generation, n.specification_digest, d.provider_kind FROM runtime_nodes n CROSS JOIN runtime_deployment d WHERE n.id=$1", nodeID).Scan(&generation, &digest, &provider); err != nil { + return err + } + var statuses []sandbox.GenerationStatus + if health.ProviderReady && provider == "microsandbox" { + statuses = []sandbox.GenerationStatus{{Generation: generation, SpecificationDigest: digest, State: "ready", Checkpoint: &sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}}} + } + return f.service.Heartbeat(ctx, nodeID, connection, epoch, health, statuses) +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/host_history_test.go b/services/core/internal/persistence/postgres/deploymentpg/host_history_test.go index ee83493e4..4b7c09e2a 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/host_history_test.go +++ b/services/core/internal/persistence/postgres/deploymentpg/host_history_test.go @@ -36,7 +36,7 @@ func TestNodeHostHistorySamplingAndDetail(t *testing.T) { now := time.Now().UTC() host := &deployment.NodeHost{EffectiveCPUCores: hostHistoryPtr(2.0), CPUUtilization: hostHistoryPtr(0.35), TotalMemoryBytes: hostHistoryPtr(int64(4096)), AvailableMemoryBytes: hostHistoryPtr(int64(1024)), AvailableDiskBytes: hostHistoryPtr(int64(8192)), ObservedAt: &now} health := deployment.NodeHealth{ProviderReady: true, Host: host} - if err := f.service.Heartbeat(t.Context(), node, connection, epoch, health); err != nil { + if err := f.heartbeat(t.Context(), node, connection, epoch, health, nil); err != nil { t.Fatal(err) } for i, want := range []int64{1, 0} { @@ -113,7 +113,7 @@ func TestNodeHostHistorySamplingAndDetail(t *testing.T) { // A disconnected node cannot create another history row, even with a fresh last observation. next := now.Add(time.Millisecond) host.ObservedAt = &next - if err := f.service.Heartbeat(t.Context(), node, connection, epoch, health); err != nil { + if err := f.heartbeat(t.Context(), node, connection, epoch, health, nil); err != nil { t.Fatal(err) } if err := f.service.DisconnectNode(t.Context(), node, connection, epoch); err != nil { @@ -136,7 +136,7 @@ func TestNodeHostHistorySamplingAndDetail(t *testing.T) { func TestNodeHostHistoryFencingAndUnknown(t *testing.T) { f, node, conn, epoch := hostHistoryNode(t) for _, at := range []time.Time{time.Now().Add(-time.Minute), time.Now().Add(time.Hour)} { - if err := f.service.Heartbeat(t.Context(), node, conn, epoch, deployment.NodeHealth{Host: &deployment.NodeHost{ObservedAt: &at}}); err != nil { + if err := f.heartbeat(t.Context(), node, conn, epoch, deployment.NodeHealth{Host: &deployment.NodeHost{ObservedAt: &at}}, nil); err != nil { t.Fatal(err) } if n, err := f.adapter.SampleHostHistory(t.Context()); err != nil || n != 0 { @@ -145,10 +145,10 @@ func TestNodeHostHistoryFencingAndUnknown(t *testing.T) { } now := time.Now().UTC() health := deployment.NodeHealth{Host: &deployment.NodeHost{ObservedAt: &now}} - if err := f.service.Heartbeat(t.Context(), node, uuid.NewString(), epoch, health); !errors.Is(err, deployment.ErrNodeCredential) { + if err := f.heartbeat(t.Context(), node, uuid.NewString(), epoch, health, nil); !errors.Is(err, deployment.ErrNodeCredential) { t.Fatal(err) } - if err := f.service.Heartbeat(t.Context(), node, conn, epoch, health); err != nil { + if err := f.heartbeat(t.Context(), node, conn, epoch, health, nil); err != nil { t.Fatal(err) } if n, err := f.adapter.SampleHostHistory(t.Context()); err != nil || n != 1 { @@ -164,7 +164,7 @@ func TestNodeHostHistoryFencingAndUnknown(t *testing.T) { {ObservedAt: &now, TotalMemoryBytes: hostHistoryPtr(int64(10)), AvailableMemoryBytes: hostHistoryPtr(int64(11))}, {ObservedAt: &now, AvailableDiskBytes: hostHistoryPtr(int64(-1))}, {ObservedAt: &now, EffectiveCPUCores: hostHistoryPtr(0.0)}, {}, } { - if err := f.service.Heartbeat(t.Context(), node, conn, epoch, deployment.NodeHealth{Host: host}); !errors.Is(err, deployment.ErrInvalidInput) { + if err := f.heartbeat(t.Context(), node, conn, epoch, deployment.NodeHealth{Host: host}, nil); !errors.Is(err, deployment.ErrInvalidInput) { t.Fatal(host, err) } } diff --git a/services/core/internal/persistence/postgres/deploymentpg/placement_demand_test.go b/services/core/internal/persistence/postgres/deploymentpg/placement_demand_test.go new file mode 100644 index 000000000..1e2f410b6 --- /dev/null +++ b/services/core/internal/persistence/postgres/deploymentpg/placement_demand_test.go @@ -0,0 +1,215 @@ +package deploymentpg_test + +import ( + "errors" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +func TestPlacementDemandAdmitsOnlyAvailableCompute(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 3, MaxRetained: 10}) + f.connect(t, node.NodeID) + keys := make([]deployment.AllocationKey, 10) + for i := range keys { + keys[i] = pendingHostedEnvironment(t, f) + if i%2 == 0 { + if _, err := f.pool.Exec(t.Context(), `UPDATE environments SET initialization='complete' WHERE id=$1`, keys[i].EnvironmentID); err != nil { + t.Fatal(err) + } + } + } + demand, _, err := f.adapter.PlacementDemand(t.Context(), deployment.PlacementDemandCursor{}) + if err != nil || len(demand) != 10 { + t.Fatal(demand, err) + } + for i, request := range demand { + _, err := changes.EnsurePlacement(t.Context(), deployment.AllocationKey{TenantID: request.TenantID, EnvironmentID: request.ID}, installation) + if i < 3 && err != nil { + t.Fatal(i, err) + } + if i >= 3 && !errors.Is(err, placement.ErrNodeUnavailable) { + t.Fatal(i, err) + } + } + var reserved, devices, allocations int + if err := f.pool.QueryRow(t.Context(), `SELECT (SELECT count(*) FROM runtime_placements WHERE released_at IS NULL),(SELECT count(*) FROM devices),(SELECT count(*) FROM runtime_allocations)`).Scan(&reserved, &devices, &allocations); err != nil { + t.Fatal(err) + } + if reserved != 3 || devices != 0 || allocations != 0 { + t.Fatal(reserved, devices, allocations) + } + // A pending node request must never enter the direct Provider's unallocated lane. + direct, err := f.adapter.UnallocatedEnvironments(t.Context(), "", "") + if err != nil || len(direct) != 0 { + t.Fatal("node demand entered direct lane", direct, err) + } + remaining, _, err := f.adapter.PlacementDemand(t.Context(), deployment.PlacementDemandCursor{}) + if err != nil || len(remaining) != 7 { + t.Fatal(remaining, err) + } + // A retried reservation consumes no additional slot or identity. + first := demand[0] + if _, err := changes.EnsurePlacement(t.Context(), deployment.AllocationKey{TenantID: first.TenantID, EnvironmentID: first.ID}, installation); err != nil { + t.Fatal(err) + } +} + +func TestPlacementDemandPagesByOriginalDemandTime(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + at := time.Now().UTC().Add(-time.Hour).Truncate(time.Microsecond) + for i := 0; i < 40; i++ { + key := pendingHostedEnvironment(t, f) + // UUIDs are random; demand order must follow durable acceptance time. + if _, err := f.pool.Exec(t.Context(), `UPDATE sessions SET created_at=$2 WHERE id=(SELECT session_id FROM environments WHERE id=$1)`, key.EnvironmentID, at.Add(time.Duration(i)*time.Second)); err != nil { + t.Fatal(err) + } + } + first, next, err := f.adapter.PlacementDemand(t.Context(), deployment.PlacementDemandCursor{}) + if err != nil || len(first) != 32 { + t.Fatal(len(first), err) + } + for i, item := range first { + if !item.At.Equal(at.Add(time.Duration(i) * time.Second)) { + t.Fatal(i, item.At) + } + } + second, end, err := f.adapter.PlacementDemand(t.Context(), next) + if err != nil || len(second) != 8 { + t.Fatal(len(second), err) + } + if end.EnvironmentID != "" || !end.Until.Equal(next.Until) { + t.Fatal("short page lost its scan horizon", end, next) + } + if !second[0].At.Equal(at.Add(32 * time.Second)) { + t.Fatal(second[0]) + } +} + +func TestFirstPlacementChecksActualGenerationHistorySupport(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + f.connect(t, node.NodeID) + key := pendingHostedEnvironment(t, f) + if _, err := f.pool.Exec(t.Context(), `UPDATE sessions SET engine='mcode' WHERE id=(SELECT session_id FROM environments WHERE id=$1)`, key.EnvironmentID); err != nil { + t.Fatal(err) + } + if _, err := changes.EnsurePlacement(t.Context(), key, installation); !errors.Is(err, placement.ErrNodeUnavailable) { + t.Fatal("unsupported Harness reserved external filesystem", err) + } + pending, _, err := f.adapter.PlacementDemand(t.Context(), deployment.PlacementDemandCursor{}) + if err != nil || len(pending) != 1 { + t.Fatal(pending, err) + } + if _, err := f.pool.Exec(t.Context(), `UPDATE sessions SET engine='codex' WHERE id=(SELECT session_id FROM environments WHERE id=$1)`, key.EnvironmentID); err != nil { + t.Fatal(err) + } + if _, err := changes.EnsurePlacement(t.Context(), key, installation); err != nil { + t.Fatal(err) + } + // Terminal state is rechecked under the Session lock, after the demand read. + another := pendingHostedEnvironment(t, f) + if _, err := f.pool.Exec(t.Context(), `UPDATE environments SET status='expired' WHERE id=$1`, another.EnvironmentID); err != nil { + t.Fatal(err) + } + if _, err := changes.EnsurePlacement(t.Context(), another, installation); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal(err) + } +} + +func pendingHostedEnvironment(t *testing.T, f fixture) deployment.AllocationKey { + t.Helper() + key := hostedEnvironment(t, f.pool) + if _, err := f.pool.Exec(t.Context(), `UPDATE environments SET initialization='pending' WHERE id=$1`, key.EnvironmentID); err != nil { + t.Fatal(err) + } + return key +} + +func TestReleasedFileWakeUsesFirstDemandAndLatestOwner(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + f.connect(t, node.NodeID) + key := hostedEnvironment(t, f.pool) + if _, err := changes.EnsurePlacement(t.Context(), key, installation); err != nil { + t.Fatal(err) + } + owner, err := changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + if err != nil { + t.Fatal(err) + } + owner = releaseRetainedFixture(t, f, changes, owner) + if err := f.service.TouchActivity(t.Context(), key.TenantID, key.EnvironmentID); err != nil { + t.Fatal(err) + } + first, _, err := f.adapter.PlacementDemand(t.Context(), deployment.PlacementDemandCursor{}) + if err != nil || len(first) != 1 || !first[0].Retained { + t.Fatal(first, err) + } + if err := f.service.TouchActivity(t.Context(), key.TenantID, key.EnvironmentID); err != nil { + t.Fatal(err) + } + replay, _, err := f.adapter.PlacementDemand(t.Context(), deployment.PlacementDemandCursor{}) + if err != nil || len(replay) != 1 || !replay[0].At.Equal(first[0].At) { + t.Fatal(replay, err) + } + if _, err := changes.EnsurePlacement(t.Context(), key, installation); err != nil { + t.Fatal(err) + } + next, err := changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + if err != nil { + t.Fatal(err) + } + if next.ID == owner.ID { + t.Fatal("wake reused released owner") + } + releaseRetainedFixture(t, f, changes, next) + idle, _, err := f.adapter.PlacementDemand(t.Context(), deployment.PlacementDemandCursor{}) + if err != nil || len(idle) != 0 { + t.Fatal("historical wake leaked into later owner", idle, err) + } +} + +func TestPlacementDemandWrapsDespiteContinuousNewArrivals(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + add := func(n int) { + for range n { + pendingHostedEnvironment(t, f) + } + } + add(64) + first, next, err := f.adapter.PlacementDemand(t.Context(), deployment.PlacementDemandCursor{}) + if err != nil || len(first) != 32 || next.Until.IsZero() { + t.Fatal(len(first), next, err) + } + horizon := next.Until + add(64) + second, next, err := f.adapter.PlacementDemand(t.Context(), next) + if err != nil || len(second) != 32 || !next.Until.Equal(horizon) { + t.Fatal(len(second), next, err) + } + add(64) + end, next, err := f.adapter.PlacementDemand(t.Context(), next) + if err != nil || len(end) != 0 || next != (deployment.PlacementDemandCursor{}) { + t.Fatal(len(end), next, err) + } + // Still-unplaced old demand is retried before the subsequent arrivals. + again, _, err := f.adapter.PlacementDemand(t.Context(), next) + if err != nil || len(again) != 32 || again[0].ID != first[0].ID { + t.Fatal(again, err) + } +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/presence_test.go b/services/core/internal/persistence/postgres/deploymentpg/presence_test.go index 52b788997..83957fb31 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/presence_test.go +++ b/services/core/internal/persistence/postgres/deploymentpg/presence_test.go @@ -246,7 +246,7 @@ func TestNodeStaleEpochCannotReplaceCurrentConnection(t *testing.T) { if err := f.service.DisconnectNode(t.Context(), node.NodeID, connection, epoch); err != nil { t.Fatal(err) } - if err := f.service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: false}); !errors.Is(err, deployment.ErrNodeCredential) { + if err := f.heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: false}, nil); !errors.Is(err, deployment.ErrNodeCredential) { t.Fatal("old Core rewrote health", err) } nodes, err := f.service.ListNodes(t.Context()) @@ -277,7 +277,7 @@ func TestNodeDiagnosticReachesListAndDetail(t *testing.T) { {reported: "dial unix /var/run/docker.sock: permission denied", want: "provider_unavailable"}, {reported: "artifacts_unavailable", want: "", ready: true}, } { - if err := f.service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: tc.ready, Diagnostic: sandbox.NodeDiagnosticCode(tc.reported)}); err != nil { + if err := f.heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: tc.ready, Diagnostic: sandbox.NodeDiagnosticCode(tc.reported)}, nil); err != nil { t.Fatal(tc.reported, err) } list, err := f.service.ListNodes(t.Context()) @@ -321,11 +321,11 @@ func TestNodeStatusUsesAuthenticatedFreshPresence(t *testing.T) { if err != nil { t.Fatal(err) } - if err := f.service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: false}); err != nil { + if err := f.heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: false}, nil); err != nil { t.Fatal(err) } status(true, false) - if err := f.service.Heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := f.heartbeat(t.Context(), node.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}, nil); err != nil { t.Fatal(err) } status(true, true) diff --git a/services/core/internal/persistence/postgres/deploymentpg/replacement_test.go b/services/core/internal/persistence/postgres/deploymentpg/replacement_test.go new file mode 100644 index 000000000..8b3a3d984 --- /dev/null +++ b/services/core/internal/persistence/postgres/deploymentpg/replacement_test.go @@ -0,0 +1,411 @@ +package deploymentpg_test + +import ( + "errors" + "sync" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/workspacefs" + "github.com/google/uuid" +) + +// releaseRetainedFixture records native cleanup completion after exercising the +// production expiry decision. It does not claim to qualify a native provider. +func releaseRetainedFixture(t *testing.T, f fixture, changes *deployment.ExecutionOperations, owner deployment.Allocation) deployment.Allocation { + t.Helper() + config := uuid.NewString() + if _, err := f.pool.Exec(t.Context(), `INSERT INTO workspace_fs_configurations(id,adapter,parameters) VALUES($1,'fixture','{}')`, config); err != nil { + t.Fatal(err) + } + if _, err := f.pool.Exec(t.Context(), `INSERT INTO environment_workspaces(object_id,environment_id,configuration_id,state,attachment) VALUES($1,$2,$3,'ready','{}') ON CONFLICT(environment_id) DO NOTHING`, uuid.NewString(), owner.EnvironmentID, config); err != nil { + t.Fatal(err) + } + if _, err := f.pool.Exec(t.Context(), `UPDATE devices SET supported_agent_kinds='[{"kind":"codex","available":true,"capabilities":{"retained_native_history":true}}]' WHERE id=$1`, owner.DeviceID); err != nil { + t.Fatal(err) + } + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET state='running',create_settled=true,compute_phase='suspended',compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID); err != nil { + t.Fatal(err) + } + owner, err := f.adapter.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + owner, err = changes.RequestCleanup(t.Context(), owner) + if err != nil { + t.Fatal(err) + } + owner, err = changes.ReleaseAllocation(t.Context(), owner) + if err != nil { + t.Fatal(err) + } + return owner +} + +func TestReplacementKeepsSessionIdentityAndFencesPriorOwner(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + firstNode := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + f.connect(t, firstNode.NodeID) + secondNode := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + f.connect(t, secondNode.NodeID) + key := hostedEnvironment(t, f.pool) + if _, err := f.pool.Exec(t.Context(), `INSERT INTO runtime_placements(environment_id,node_id,deployment_generation) VALUES($1,$2,$3)`, key.EnvironmentID, firstNode.NodeID, view.Generation); err != nil { + t.Fatal(err) + } + first, err := changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + if err != nil { + t.Fatal(err) + } + if _, err := f.pool.Exec(t.Context(), `UPDATE session_devices SET native_session_id='retained-native-session' WHERE device_id=$1`, first.DeviceID); err != nil { + t.Fatal(err) + } + first = releaseRetainedFixture(t, f, changes, first) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, firstNode.NodeID); err != nil { + t.Fatal(err) + } + reserved, err := changes.EnsurePlacement(t.Context(), key, installation) + if err != nil || reserved.NodeID != secondNode.NodeID { + t.Fatal(reserved, err) + } + // Replayed old cleanup must not release the new reservation, even before Create. + if _, err := changes.ReleaseAllocation(t.Context(), first); err != nil { + t.Fatal(err) + } + var released bool + if err := f.pool.QueryRow(t.Context(), `SELECT released_at IS NOT NULL FROM runtime_placements WHERE environment_id=$1`, key.EnvironmentID).Scan(&released); err != nil || released { + t.Fatal("late release removed reservation", released, err) + } + owners := make([]deployment.Allocation, 6) + errs := make([]error, len(owners)) + var wg sync.WaitGroup + for i := range owners { + wg.Go(func() { + owners[i], errs[i] = changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + }) + } + wg.Wait() + fresh := 0 + for i, owner := range owners { + if errs[i] != nil || owner.ID != owners[0].ID || owner.NodeID != secondNode.NodeID { + t.Fatal(owner, errs[i]) + } + if !owner.Replayed { + fresh++ + } + } + second := owners[0] + if fresh != 1 || second.ID == first.ID || second.DeviceID == first.DeviceID { + t.Fatal("replacement reused identity", fresh, first, second) + } + var device, native string + if err := f.pool.QueryRow(t.Context(), `SELECT device_id,native_session_id FROM session_devices WHERE session_id=$1`, second.SessionID).Scan(&device, &native); err != nil || device != second.DeviceID || native != "retained-native-session" { + t.Fatal(device, native, err) + } + for name, write := range map[string]func() (deployment.Allocation, error){"cleanup": func() (deployment.Allocation, error) { return changes.RequestCleanup(t.Context(), first) }, "release": func() (deployment.Allocation, error) { return changes.ReleaseAllocation(t.Context(), first) }, "observe": func() (deployment.Allocation, error) { return changes.ObserveRunning(t.Context(), first) }} { + if _, err := write(); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal(name, "accepted stale owner", err) + } + } + var history, current, authority int + if err := f.pool.QueryRow(t.Context(), `SELECT count(*),count(*) FILTER(WHERE state<>'released') FROM runtime_allocations WHERE environment_id=$1`, key.EnvironmentID).Scan(&history, ¤t); err != nil || history != 2 || current != 1 { + t.Fatal(history, current, err) + } + if err := f.pool.QueryRow(t.Context(), `SELECT count(*) FROM runtime_device_authority WHERE id=$1`, first.DeviceID).Scan(&authority); err != nil || authority != 0 { + t.Fatal("old credential still authorized", authority, err) + } + var oldNode string + if err := f.pool.QueryRow(t.Context(), `SELECT node_id FROM runtime_allocations WHERE id=$1`, first.ID).Scan(&oldNode); err != nil || oldNode != firstNode.NodeID { + t.Fatal("historical node changed", oldNode, err) + } + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil { + t.Fatal(err) + } + for _, node := range nodes { + if node.ID == secondNode.NodeID && (node.Active != 1 || node.Retained != 1) { + t.Fatal("history counted as compute", node) + } + } + // The database independently rejects a second live owner, even outside the domain transaction. + otherDevice := uuid.NewString() + if _, err := f.pool.Exec(t.Context(), `INSERT INTO devices(id,tenant_id,name,credential_hash) VALUES($1,$2,'fixture',$3)`, otherDevice, key.TenantID, credentialHash()); err != nil { + t.Fatal(err) + } + if _, err := f.pool.Exec(t.Context(), `INSERT INTO runtime_allocations(id,environment_id,device_id,provider_key,node_id,deployment_generation) VALUES($1,$2,$3,$4,$5,$6)`, uuid.NewString(), key.EnvironmentID, otherDevice, installation, secondNode.NodeID, view.Generation); err == nil { + t.Fatal("database accepted two current allocations") + } +} + +func TestReplacementRefusesUnsettledOwnershipAndUnknownHistory(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + f.connect(t, node.NodeID) + key := hostedEnvironment(t, f.pool) + if _, err := f.pool.Exec(t.Context(), `INSERT INTO runtime_placements(environment_id,node_id,deployment_generation) VALUES($1,$2,$3)`, key.EnvironmentID, node.NodeID, view.Generation); err != nil { + t.Fatal(err) + } + owner, err := changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + if err != nil { + t.Fatal(err) + } + if _, err := changes.EnsurePlacement(t.Context(), key, installation); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("unknown writer admitted replacement", err) + } + replay, err := changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + if err != nil || !replay.Replayed || replay.ID != owner.ID { + t.Fatal("unknown Create was replayed", replay, err) + } + owner = releaseRetainedFixture(t, f, changes, owner) + for _, raw := range []string{`[]`, `[{"kind":"codex","available":true,"capabilities":{}}]`, `[{"kind":"codex","available":false,"capabilities":{"retained_native_history":true}}]`, `[{"kind":"different","available":true,"capabilities":{"retained_native_history":true}}]`} { + if _, err := f.pool.Exec(t.Context(), `UPDATE devices SET supported_agent_kinds=$2 WHERE id=$1`, owner.DeviceID, raw); err != nil { + t.Fatal(err) + } + if allowed, err := f.adapter.RetainedNativeHistory(t.Context(), key); err != nil || allowed { + t.Fatal("unknown history admitted", raw, allowed, err) + } + if _, err := changes.EnsurePlacement(t.Context(), key, installation); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("unknown history reserved replacement", err) + } + } + if _, err := f.pool.Exec(t.Context(), `UPDATE devices SET supported_agent_kinds='[{"kind":"codex","available":true,"capabilities":{"retained_native_history":true}}]',revoked_at=NULL WHERE id=$1`, owner.DeviceID); err != nil { + t.Fatal(err) + } + if _, err := changes.EnsurePlacement(t.Context(), key, installation); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("old device still had authority", err) + } +} + +func TestTenRetainedWorkspacesShareThreeComputeSlots(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 3, MaxRetained: 3}) + f.connect(t, node.NodeID) + keys := make([]deployment.AllocationKey, 10) + for i := range keys { + key := hostedEnvironment(t, f.pool) + keys[i] = key + if _, err := f.pool.Exec(t.Context(), `INSERT INTO runtime_placements(environment_id,node_id,deployment_generation) VALUES($1,$2,$3)`, key.EnvironmentID, node.NodeID, view.Generation); err != nil { + t.Fatal(err) + } + owner, err := changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + if err != nil { + t.Fatal(err) + } + releaseRetainedFixture(t, f, changes, owner) + } + check := func(active, retained int64) { + t.Helper() + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil || len(nodes) != 1 { + t.Fatal(nodes, err) + } + if nodes[0].Active != active || nodes[0].Retained != retained { + t.Fatal("incorrect compute accounting", nodes[0]) + } + var objects int + if err := f.pool.QueryRow(t.Context(), `SELECT count(*) FROM environment_workspaces WHERE state='ready'`).Scan(&objects); err != nil || objects != 10 { + t.Fatal("retained filesystem count", objects, err) + } + } + check(0, 0) + owners := make([]deployment.Allocation, 3) + for i := range owners { + if _, err := changes.EnsurePlacement(t.Context(), keys[i], installation); err != nil { + t.Fatal(err) + } + owner, err := changes.ReserveAllocation(t.Context(), keys[i], installation, credentialHash()) + if err != nil { + t.Fatal(err) + } + owners[i] = owner + } + check(3, 3) + if _, err := changes.EnsurePlacement(t.Context(), keys[3], installation); err == nil { + t.Fatal("fourth demand exceeded capacity") + } + var count int + if err := f.pool.QueryRow(t.Context(), `SELECT count(*) FROM runtime_allocations WHERE environment_id=$1`, keys[3].EnvironmentID).Scan(&count); err != nil || count != 1 { + t.Fatal("rejected demand created allocation", count, err) + } + check(3, 3) + releaseRetainedFixture(t, f, changes, owners[0]) + check(2, 2) + if _, err := changes.EnsurePlacement(t.Context(), keys[3], installation); err != nil { + t.Fatal(err) + } + // A committed reservation is itself retry demand for live file access. + pending, err := f.adapter.UnallocatedEnvironments(t.Context(), node.NodeID, "") + if err != nil || len(pending) != 1 || pending[0].ID != keys[3].EnvironmentID { + t.Fatal("reservation lost without pending input", pending, err) + } + if _, err := changes.ReserveAllocation(t.Context(), keys[3], installation, credentialHash()); err != nil { + t.Fatal(err) + } + check(3, 3) +} + +func TestResetArchivesRetainedSessionBeforeClearingDeployment(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + f.connect(t, node.NodeID) + retained := make([]deployment.Allocation, 37) + for i := range retained { + key := hostedEnvironment(t, f.pool) + if _, err := f.pool.Exec(t.Context(), `INSERT INTO runtime_placements(environment_id,node_id,deployment_generation) VALUES($1,$2,$3)`, key.EnvironmentID, node.NodeID, view.Generation); err != nil { + t.Fatal(err) + } + owner, err := changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + if err != nil { + t.Fatal(err) + } + retained[i] = releaseRetainedFixture(t, f, changes, owner) + } + target := retained[0] + // Only the exact latest ownership in this reset's installation and boundary counts. + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET provider_key=$2 WHERE id=$1`, retained[1].ID, uuid.NewString()); err != nil { + t.Fatal(err) + } + futureDevice := uuid.NewString() + if _, err := f.pool.Exec(t.Context(), `INSERT INTO devices(id,tenant_id,name,credential_hash,environment_id,revoked_at) VALUES($1,$2,'future', $3,$4,clock_timestamp())`, futureDevice, retained[2].TenantID, credentialHash(), retained[2].EnvironmentID); err != nil { + t.Fatal(err) + } + if _, err := f.pool.Exec(t.Context(), `INSERT INTO runtime_allocations(id,environment_id,device_id,provider_key,node_id,deployment_generation,state,create_settled,released_at) VALUES($1,$2,$3,$4,$5,$6,'released',true,clock_timestamp())`, uuid.NewString(), retained[2].EnvironmentID, futureDevice, installation, node.NodeID, view.Generation+1); err != nil { + t.Fatal(err) + } + if _, err := f.pool.Exec(t.Context(), `UPDATE environments SET status='expired' WHERE id=$1`, retained[3].EnvironmentID); err != nil { + t.Fatal(err) + } + if err := changes.StartReset(admin(t), installation, deployment.ResetRequest{ExpectedGeneration: view.Generation, Clear: deployment.ResetForce}); err != nil { + t.Fatal(err) + } + started, err := f.service.View(t.Context()) + if err != nil || started.Reset == nil { + t.Fatal(started, err) + } + if _, err := f.pool.Exec(t.Context(), `UPDATE sessions SET created_at=$2 WHERE id=$1`, retained[4].SessionID, started.Reset.RequestedAt.Add(time.Second)); err != nil { + t.Fatal(err) + } + candidates, err := f.adapter.ResetSessions(t.Context(), "", true) + if err != nil || len(candidates) != 32 { + t.Fatal("reset scope", candidates, err) + } + snapshot, err := f.service.View(t.Context()) + if err != nil || snapshot.Resources.Pending != 33 { + t.Fatal("retained reset work missing from completion gate", snapshot.Resources, err) + } + for _, candidate := range candidates { + for _, excluded := range retained[1:5] { + if candidate.SessionID == excluded.SessionID { + t.Fatal("out-of-scope history entered reset", candidate) + } + } + } + var inUse *deployment.InUseError + if _, err := changes.CompleteReset(t.Context(), installation, view.Generation, started.Reset.RequestedAt); !errors.As(err, &inUse) { + t.Fatal("reset completed before archive", err) + } + archive := func(candidates []deployment.ResetSession) { + t.Helper() + for _, candidate := range candidates { + if _, err := f.pool.Exec(t.Context(), `INSERT INTO execution_project_scopes(tenant_id,organization_id,project_id) VALUES($1,$2,$3)`, candidate.TenantID, uuid.NewString(), uuid.NewString()); err != nil { + t.Fatal(err) + } + if _, err := f.pool.Exec(t.Context(), `INSERT INTO projects(id,name,tenant_id,subject_kind,subject_id) VALUES($1,'retained',$2,'service_account','fixture')`, uuid.NewString(), candidate.TenantID); err != nil { + t.Fatal(err) + } + if _, err := changes.ArchiveResetSession(t.Context(), candidate.TenantID, candidate.SessionID, view.Generation, started.Reset.RequestedAt); err != nil { + t.Fatal(err) + } + } + } + archive(candidates) + remaining, err := f.adapter.ResetSessions(t.Context(), "", true) + if err != nil || len(remaining) != 1 { + t.Fatal("second reset page", remaining, err) + } + snapshot, err = f.service.View(t.Context()) + if err != nil || snapshot.Resources.Pending != 1 { + t.Fatal("second page did not block completion", snapshot.Resources, err) + } + if _, err := changes.CompleteReset(t.Context(), installation, view.Generation, started.Reset.RequestedAt); !errors.As(err, &inUse) { + t.Fatal("reset skipped second page", err) + } + archive(remaining) + committed, err := changes.CompleteReset(t.Context(), installation, view.Generation, started.Reset.RequestedAt) + if err != nil { + t.Fatal(err) + } + next, err := changes.Initialize(admin(t), installation, sandbox.Selection{Provider: "microsandbox", ExpectedGeneration: committed, DeploymentSpec: retainedSpecification()}) + if err != nil { + t.Fatal(err) + } + nextNode := f.enroll(t, next, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + f.connect(t, nextNode.NodeID) + if _, err := changes.EnsurePlacement(t.Context(), target.Key(), installation); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("reset Session revived in new deployment", err) + } + var status, fsState string + if err := f.pool.QueryRow(t.Context(), `SELECT e.status,w.state FROM environments e JOIN environment_workspaces w ON w.environment_id=e.id WHERE e.id=$1`, target.EnvironmentID).Scan(&status, &fsState); err != nil || status != "expired" || fsState != "ready" { + t.Fatal("archive lost files or kept Session executable", status, fsState, err) + } +} + +func retainedSpecification() sandbox.DeploymentSpec { + spec := testSpecification("microsandbox") + spec.Resources.RootDiskMiB = 8192 + spec.Resources.EnvironmentDiskMiB = 0 + spec.Workspace = &workspacefs.Declaration{Attachment: workspacefs.AttachmentHostDirectory, UserXAttr: true} + return spec +} + +func TestReplacementWaitsForCompatibleReadyGenerationWithoutReservingOwnedNode(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, owned := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: testSpecification("microsandbox")}) + oldNode := f.enroll(t, owned, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + f.connect(t, oldNode.NodeID) + external, err := changes.Update(admin(t), installation, sandbox.Selection{Provider: "microsandbox", ExpectedGeneration: owned.Generation, DeploymentSpec: retainedSpecification()}) + if err != nil { + t.Fatal(err) + } + newNode := f.enroll(t, external, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + f.connect(t, newNode.NodeID) + key := hostedEnvironment(t, f.pool) + if _, err = f.pool.Exec(t.Context(), `INSERT INTO runtime_placements(environment_id,node_id,deployment_generation) VALUES($1,$2,$3)`, key.EnvironmentID, newNode.NodeID, external.Generation); err != nil { + t.Fatal(err) + } + owner, err := changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + if err != nil { + t.Fatal(err) + } + releaseRetainedFixture(t, f, changes, owner) + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, newNode.NodeID); err != nil { + t.Fatal(err) + } + if _, err = changes.EnsurePlacement(t.Context(), key, installation); !errors.Is(err, placement.ErrNodeUnavailable) { + t.Fatal("owned generation admitted retained FS", err) + } + var released bool + if err = f.pool.QueryRow(t.Context(), `SELECT released_at IS NOT NULL FROM runtime_placements WHERE environment_id=$1`, key.EnvironmentID).Scan(&released); err != nil || !released { + t.Fatal("incompatible placement consumed capacity", released, err) + } + f.connect(t, newNode.NodeID) + reserved, err := changes.EnsurePlacement(t.Context(), key, installation) + if err != nil || reserved.NodeID != newNode.NodeID { + t.Fatal("compatible capacity did not recover demand", reserved, err) + } + fresh, err := changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + if err != nil || fresh.ID == owner.ID || fresh.NodeID != newNode.NodeID { + t.Fatal(fresh, err) + } +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/retention_pressure_test.go b/services/core/internal/persistence/postgres/deploymentpg/retention_pressure_test.go new file mode 100644 index 000000000..ac6864145 --- /dev/null +++ b/services/core/internal/persistence/postgres/deploymentpg/retention_pressure_test.go @@ -0,0 +1,112 @@ +package deploymentpg_test + +import ( + "encoding/json" + "errors" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/google/uuid" +) + +func qualifyPressureRetention(t *testing.T, f fixture, owner deployment.Allocation) deployment.Allocation { + t.Helper() + config := uuid.NewString() + for _, statement := range []struct { + sql string + args []any + }{ + {`INSERT INTO workspace_fs_configurations(id,adapter,parameters) VALUES($1,'fixture','{}')`, []any{config}}, + {`INSERT INTO environment_workspaces(object_id,environment_id,configuration_id,state,attachment) VALUES($1,$2,$3,'ready','{}')`, []any{uuid.NewString(), owner.EnvironmentID, config}}, + {`UPDATE devices SET supported_agent_kinds='[{"kind":"codex","available":true,"capabilities":{"retained_native_history":true}}]' WHERE id=$1`, []any{owner.DeviceID}}, + {`UPDATE runtime_allocations SET compute_phase='suspended',compute_state='{}',compute_retained_until=clock_timestamp()+interval '24 hours' WHERE id=$1`, []any{owner.ID}}, + } { + if _, err := f.pool.Exec(t.Context(), statement.sql, statement.args...); err != nil { + t.Fatal(err) + } + } + current, err := f.adapter.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + return current +} + +func TestRetentionPressureKeepsHistoryUntilOrdinaryCleanupSettles(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 3, MaxRetained: 3}) + f.connect(t, node.NodeID) + owner := qualifyPressureRetention(t, f, runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation)) + for range 2 { + qualifyPressureRetention(t, f, runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation)) + } + waiting := pendingPressureDemand(t, f) + for range 6 { + pendingPressureDemand(t, f) + } + if _, err := changes.EnsurePlacement(t.Context(), waiting, installation); !errors.Is(err, placement.ErrNodeUnavailable) { + t.Fatal("reserved despite retained capacity limit") + } + current, err := f.adapter.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || current.Expired || !current.ComputeRetainedUntil.Equal(*owner.ComputeRetainedUntil) { + t.Fatal("pressure shortened retention", current, err) + } + // Advance the fixture clock past the authored deadline, not because of demand. + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_retained_until=clock_timestamp()-interval '1 second' WHERE id=$1`, owner.ID); err != nil { + t.Fatal(err) + } + // The database deadline is only an intent. It is not a cleanup receipt. + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil || nodes[0].Retained != 3 || nodes[0].Active != 0 { + t.Fatal("released capacity before cleanup", nodes, err) + } + if _, err = changes.EnsurePlacement(t.Context(), waiting, installation); !errors.Is(err, placement.ErrNodeUnavailable) { + t.Fatal("reserved before cleanup") + } + current, err = f.adapter.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil || !current.Expired { + t.Fatal("database expiry", current, err) + } + pending, err := changes.RequestCleanup(t.Context(), current) + if err != nil { + t.Fatal(err) + } + var status, filesystem string + if err = f.pool.QueryRow(t.Context(), `SELECT e.status,w.state FROM environments e JOIN environment_workspaces w ON w.environment_id=e.id WHERE e.id=$1`, owner.EnvironmentID).Scan(&status, &filesystem); err != nil || status == "expired" || status == "failed" || filesystem != "ready" { + t.Fatal("lost retained environment", status, filesystem, err) + } + // This fixture supplies the native cleanup completion that a real lifecycle + // must obtain before invoking ReleaseAllocation. + if _, err = changes.ReleaseAllocation(t.Context(), pending); err != nil { + t.Fatal(err) + } + if _, err = changes.EnsurePlacement(t.Context(), waiting, installation); err != nil { + t.Fatal("waiting demand did not progress", err) + } +} + +func TestPressurePreservesRetainableOwnerWhenRetainedSlotsAreFull(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 1}) + f.connect(t, node.NodeID) + owner := qualifyPressureRetention(t, f, runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation)) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='running',compute_retained_until=NULL WHERE id=$1`, owner.ID); err != nil { + t.Fatal(err) + } + owner, err := f.adapter.EnvironmentAllocation(t.Context(), owner.Key()) + if err != nil { + t.Fatal(err) + } + pendingPressureDemand(t, f) + until := time.Now().Add(24 * time.Hour) + if _, err = changes.SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute); !errors.Is(err, deployment.ErrNotIdle) { + t.Fatal("suspended without usable capacity", err) + } +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/setup_test.go b/services/core/internal/persistence/postgres/deploymentpg/setup_test.go index 6fee76efc..972395dd1 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/setup_test.go +++ b/services/core/internal/persistence/postgres/deploymentpg/setup_test.go @@ -6,6 +6,7 @@ import ( "encoding/hex" "errors" "fmt" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "sync" "testing" @@ -56,7 +57,7 @@ func setupClosedExecution(t *testing.T, f fixture) *deployment.ExecutionOperatio t.Fatal(err) } awaitRelease() - operations, err := deployment.NewExecutionOperations(f.service, deploymentpg.NewExecution(lease, f.cipher)) + operations, err := deployment.NewExecutionOperations(f.service, deploymentpg.NewExecution(lease, f.cipher), engine.Catalog{}) if err != nil { t.Fatal(err) } diff --git a/services/core/internal/persistence/postgres/deploymentpg/store.go b/services/core/internal/persistence/postgres/deploymentpg/store.go index 612af96bc..e4aee9e74 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/store.go +++ b/services/core/internal/persistence/postgres/deploymentpg/store.go @@ -116,7 +116,7 @@ func (s *Store) Allocation(ctx context.Context, ref sandbox.Reference) (deployme var result deployment.AllocationRecord err = s.pool.Snapshot(ctx, func(ctx context.Context, tx pgx.Tx) error { q := sqlc.New(tx) - row, err := q.GetRuntimeAllocation(ctx, sqlc.GetRuntimeAllocationParams{TenantID: tenant, EnvironmentID: environment}) + row, err := q.GetLatestRuntimeAllocation(ctx, sqlc.GetLatestRuntimeAllocationParams{TenantID: tenant, EnvironmentID: environment}) if errors.Is(err, pgx.ErrNoRows) { return deployment.ErrNotFound } diff --git a/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure.go b/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure.go new file mode 100644 index 000000000..aa1d7f0fd --- /dev/null +++ b/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure.go @@ -0,0 +1,131 @@ +package deploymentpg + +import ( + "context" + "encoding/json" + + "github.com/jackc/pgx/v5/pgtype" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/db/sqlc" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/placementpg" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +func (t *allocationTx) LoadGenerationSpecification(generation uint64) (deployment.GenerationSpecification, error) { + return (unit{ctx: t.ctx, q: t.q}).LoadGenerationSpecification(generation) +} + +func (t *allocationTx) LoadSuspensionDemand(current deployment.Allocation) (deployment.SuspensionDemand, error) { + var result deployment.SuspensionDemand + var err error + result.Deployment, err = placementpg.LockDeployment(t.ctx, t.q) + if err != nil { + return result, err + } + inFlight, err := t.q.ListRuntimeSuspensionInFlightNodes(t.ctx) + if err != nil || result.Deployment.Resetting { + return result, err + } + for _, node := range inFlight { + result.InFlightNodes = append(result.InFlightNodes, uuidString(node)) + } + result.Nodes, err = placementpg.LoadNodes(t.ctx, t.q) + if err != nil { + return result, err + } + restores, err := t.q.ListWaitingCheckpointRestores(t.ctx) + if err != nil { + return result, err + } + byGeneration := map[int64][]placement.CheckpointNode{} + for _, restore := range restores { + var compatibility sandbox.CheckpointCompatibility + if err := json.Unmarshal(restore.Checkpoint, &compatibility); err != nil { + return result, err + } + if compatibility.Validate() != nil { + continue + } + nodes, ok := byGeneration[restore.DeploymentGeneration.Int64] + if !ok { + nodes, err = placementpg.LoadCheckpointNodes(t.ctx, t.q, uint64(restore.DeploymentGeneration.Int64), result.Nodes) + if err != nil { + return result, err + } + byGeneration[restore.DeploymentGeneration.Int64] = nodes + } + result.CheckpointRestores = append(result.CheckpointRestores, deployment.CheckpointRestoreDemand{SourceNode: uuidString(restore.NodeID), Compatibility: compatibility, Nodes: nodes}) + } + return result, nil +} + +func (t *allocationTx) PlacementDemand(after deployment.PlacementDemandCursor) ([]deployment.PlacementDemand, deployment.PlacementDemandCursor, error) { + rows, next, err := loadPlacementDemand(t.ctx, t.q, after) + if err != nil { + return nil, deployment.PlacementDemandCursor{}, err + } + qualified := rows[:0] + for _, row := range rows { + if row.Retained { + tenant, environment, err := allocationKey(deployment.AllocationKey{TenantID: row.TenantID, EnvironmentID: row.ID}) + if err != nil { + return nil, next, err + } + owner, found, err := findAllocation(t.ctx, t.q, tenant, environment) + if err != nil { + return nil, next, err + } + if !found { + continue + } + id, err := parseID(owner.ID) + if err != nil { + return nil, next, err + } + replaceable, err := t.q.CanReplaceRuntimeAllocation(t.ctx, id) + if err != nil { + return nil, next, err + } + if !replaceable { + continue + } + retained, err := t.q.CanRetainRuntimeEnvironment(t.ctx, id) + if err != nil { + return nil, next, err + } + if !retained { + continue + } + } + qualified = append(qualified, row) + } + return qualified, next, nil +} + +func loadPlacementDemand(ctx context.Context, q *sqlc.Queries, after deployment.PlacementDemandCursor) ([]deployment.PlacementDemand, deployment.PlacementDemandCursor, error) { + var next deployment.PlacementDemandCursor + id, err := cursor(after.EnvironmentID) + if err != nil { + return nil, next, err + } + rows, err := q.ListPlacementDemand(ctx, sqlc.ListPlacementDemandParams{AfterID: id, AfterTime: pgtype.Timestamptz{Time: after.At, Valid: true}, UntilTime: pgtype.Timestamptz{Time: after.Until, Valid: !after.Until.IsZero()}}) + if err != nil { + return nil, next, err + } + result := make([]deployment.PlacementDemand, 0, len(rows)) + for _, row := range rows { + result = append(result, deployment.PlacementDemand{UnallocatedEnvironment: deployment.UnallocatedEnvironment{ID: uuidString(row.ID), TenantID: uuidString(row.TenantID)}, Engine: row.Engine, Retained: row.Retained, At: row.DemandedAt.Time}) + } + // Even a short final page carries its database horizon, so a caller + // yielding partway through can resume without admitting new arrivals. + if len(rows) > 0 { + next.Until = rows[0].ScanUntil.Time + } + if len(rows) == 32 { + last := rows[len(rows)-1] + next = deployment.PlacementDemandCursor{At: last.DemandedAt.Time, EnvironmentID: uuidString(last.ID), Until: last.ScanUntil.Time} + } + return result, next, nil +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure_test.go b/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure_test.go new file mode 100644 index 000000000..67acb835e --- /dev/null +++ b/services/core/internal/persistence/postgres/deploymentpg/suspension_pressure_test.go @@ -0,0 +1,311 @@ +package deploymentpg_test + +import ( + "encoding/json" + "errors" + "sync" + "testing" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + "github.com/google/uuid" +) + +func runningPressureOwner(t *testing.T, f fixture, changes *deployment.ExecutionOperations, installation, node string, generation uint64) deployment.Allocation { + t.Helper() + key := hostedEnvironment(t, f.pool) + if _, err := f.pool.Exec(t.Context(), `INSERT INTO runtime_placements(environment_id,node_id,deployment_generation) VALUES($1,$2,$3)`, key.EnvironmentID, node, generation); err != nil { + t.Fatal(err) + } + owner, err := changes.ReserveAllocation(t.Context(), key, installation, credentialHash()) + if err != nil { + t.Fatal(err) + } + if _, err = f.pool.Exec(t.Context(), `UPDATE environments SET initialization='complete' WHERE id=$1`, key.EnvironmentID); err != nil { + t.Fatal(err) + } + if _, err = f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET state='running',create_settled=true,compute_phase='running',compute_activity_at=clock_timestamp()-interval '1 minute',compute_phase_changed_at=clock_timestamp()-interval '1 minute' WHERE id=$1`, owner.ID); err != nil { + t.Fatal(err) + } + owner, err = f.adapter.EnvironmentAllocation(t.Context(), key) + if err != nil { + t.Fatal(err) + } + return owner +} + +func TestPressureSuspensionRequiresUsefulCapacity(t *testing.T) { + for _, test := range []struct { + name string + retained int + demand, freeNode bool + want error + }{ + {"no demand", 3, false, false, deployment.ErrNotIdle}, + {"retained full", 1, true, false, deployment.ErrNotIdle}, + {"other node free", 3, true, true, deployment.ErrNotIdle}, + {"waiting first session", 3, true, false, nil}, + } { + t.Run(test.name, func(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: testSpecification("microsandbox")}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: test.retained}) + f.connect(t, node.NodeID) + owner := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) + if test.demand { + pendingPressureDemand(t, f) + } + if test.freeNode { + other := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 3}) + f.connect(t, other.NodeID) + } + until := time.Now().Add(time.Hour) + _, err := changes.SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute) + if !errors.Is(err, test.want) { + t.Fatalf("quiesce: %v, want %v", err, test.want) + } + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil { + t.Fatal(err) + } + for _, n := range nodes { + if n.ID == node.NodeID && (n.Active != 1 || n.Retained != 1) { + t.Fatalf("unconfirmed suspension freed capacity: %#v", n) + } + } + }) + } +} + +func TestPressureSuspensionSerializesAcrossNodes(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: testSpecification("microsandbox")}) + owners := make([]deployment.Allocation, 2) + for i := range owners { + node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 3}) + f.connect(t, node.NodeID) + owners[i] = runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) + } + pendingPressureDemand(t, f) + var wg sync.WaitGroup + errs := make([]error, len(owners)) + for i, owner := range owners { + wg.Go(func() { + until := time.Now().Add(time.Hour) + _, errs[i] = changes.SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute) + }) + } + wg.Wait() + success := 0 + for _, err := range errs { + if err == nil { + success++ + } else if !errors.Is(err, deployment.ErrNotIdle) { + t.Fatal(err) + } + } + if success != 1 { + t.Fatalf("one waiting Session started %d suspensions: %v", success, errs) + } + var phases int + if err := f.pool.QueryRow(t.Context(), `SELECT count(*) FROM runtime_allocations WHERE compute_phase='quiescing'`).Scan(&phases); err != nil || phases != 1 { + t.Fatal(phases, err) + } +} + +func pendingPressureDemand(t *testing.T, f fixture) deployment.AllocationKey { + t.Helper() + key := hostedEnvironment(t, f.pool) + if _, err := f.pool.Exec(t.Context(), `UPDATE environments SET initialization='pending' WHERE id=$1`, key.EnvironmentID); err != nil { + t.Fatal(err) + } + return key +} + +func TestPressureSuspensionServesRestoreOnItsOriginalNode(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: testSpecification("microsandbox")}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + f.connect(t, node.NodeID) + waiting := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase='suspended',compute_state='{"protocol_version":"1","retained":{"Compatibility":{"artifact_domain":"fixture-store","execution_class":"fixture-runtime"}}}',compute_retained_until=clock_timestamp()+interval '1 hour',compute_wake_requested=true WHERE id=$1`, waiting.ID); err != nil { + t.Fatal(err) + } + owner := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) + // Free capacity in another artifact domain cannot restore this checkpoint. + other := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + f.connect(t, other.NodeID) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_node_generation_status SET checkpoint='{"artifact_domain":"other","execution_class":"fixture-runtime"}' WHERE node_id=$1`, other.NodeID); err != nil { + t.Fatal(err) + } + until := time.Now().Add(time.Hour) + if _, err := changes.SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute); err != nil { + t.Fatal(err) + } + var wake bool + if err := f.pool.QueryRow(t.Context(), `SELECT compute_wake_requested FROM runtime_allocations WHERE id=$1`, waiting.ID).Scan(&wake); err != nil || !wake { + t.Fatal("pressure consumed the waiting wake", wake, err) + } +} + +func TestPressureDemandScansPastIncompatiblePage(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + capacity := deployment.Capacity{MaxActive: 1, MaxRetained: 2} + node := f.enroll(t, view, capacity) + f.connect(t, node.NodeID) + owner := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) + for range 32 { + key := pendingPressureDemand(t, f) + if _, err := f.pool.Exec(t.Context(), `UPDATE sessions SET engine='mcode' WHERE id=(SELECT session_id FROM environments WHERE id=$1)`, key.EnvironmentID); err != nil { + t.Fatal(err) + } + } + pendingPressureDemand(t, f) + until := time.Now().Add(time.Hour) + result, err := changes.SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute) + if err != nil { + t.Fatal(err) + } + if result.ComputePhase != "quiescing" { + t.Fatal("later compatible demand did not suspend compute") + } +} + +func TestPressureDemandRejectsUnqualifiedRetainedReceipt(t *testing.T) { + for _, badReceipt := range []string{"history", "credential"} { + t.Run(badReceipt, func(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: retainedSpecification()}) + node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + f.connect(t, node.NodeID) + count := 1 + if badReceipt == "history" { + count = 32 + } + for range count { + old := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) + old = releaseRetainedFixture(t, f, changes, old) + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_wake_requested=true WHERE id=$1`, old.ID); err != nil { + t.Fatal(err) + } + statement := `UPDATE devices SET supported_agent_kinds='[]' WHERE id=$1` + if badReceipt == "credential" { + statement = `UPDATE devices SET revoked_at=NULL WHERE id=$1` + } + if _, err := f.pool.Exec(t.Context(), statement, old.DeviceID); err != nil { + t.Fatal(err) + } + } + owner := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) + until := time.Now().Add(time.Hour) + if _, err := changes.SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute); !errors.Is(err, deployment.ErrNotIdle) { + t.Fatal("unqualified receipt suspended healthy owner", err) + } + // An empty qualified page must not stop the cursor before a later + // valid request, even though the raw first page was nonempty. + pendingPressureDemand(t, f) + if _, err := changes.SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute); err != nil { + t.Fatal("later valid demand was hidden", err) + } + }) + } +} + +func TestPressureReclamationIgnoresUnavailableNodeReceipts(t *testing.T) { + for _, obstruction := range []string{"offline expired", "offline quiescing", "offline suspending", "different address", "incompatible generation"} { + t.Run(obstruction, func(t *testing.T) { + f := newFixture(t) + changes, _ := f.execution(t) + spec := retainedSpecification() + if obstruction == "incompatible generation" { + spec = testSpecification("microsandbox") + } + installation, view := f.initialize(t, changes, sandbox.Selection{Provider: "microsandbox", DeploymentSpec: spec}) + badNode := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + f.connect(t, badNode.NodeID) + stuck := runningPressureOwner(t, f, changes, installation, badNode.NodeID, view.Generation) + phase := "quiescing" + if obstruction == "offline suspending" { + phase = "suspending" + } + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_phase=$2,compute_retained_until=clock_timestamp()+interval '1 hour' WHERE id=$1`, stuck.ID, phase); err != nil { + t.Fatal(err) + } + switch obstruction { + case "offline expired", "offline quiescing", "offline suspending": + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1`, badNode.NodeID); err != nil { + t.Fatal(err) + } + if obstruction == "offline expired" { + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_allocations SET compute_retained_until=clock_timestamp()-interval '1 minute' WHERE id=$1`, stuck.ID); err != nil { + t.Fatal(err) + } + } + case "different address": + if _, err := f.pool.Exec(t.Context(), `UPDATE runtime_nodes SET core_url='https://other.example' WHERE id=$1`, badNode.NodeID); err != nil { + t.Fatal(err) + } + case "incompatible generation": + var err error + view, err = changes.Update(admin(t), installation, sandbox.Selection{Provider: "microsandbox", ExpectedGeneration: view.Generation, DeploymentSpec: retainedSpecification()}) + if err != nil { + t.Fatal(err) + } + } + node := f.enroll(t, view, deployment.Capacity{MaxActive: 1, MaxRetained: 2}) + f.connect(t, node.NodeID) + var waiting deployment.AllocationKey + if obstruction == "incompatible generation" { + old := releaseRetainedFixture(t, f, changes, runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation)) + waiting = old.Key() + } else { + waiting = pendingPressureDemand(t, f) + } + // A real pending reservation supplies demand and retains its original deadline. + if _, err := f.pool.Exec(t.Context(), `INSERT INTO environment_input_reservations(id,session_id,idempotency_key,batch,created_at,deadline) SELECT $1,session_id,'pressure','[{}]',clock_timestamp(),clock_timestamp()+interval '5 minutes' FROM environments WHERE id=$2`, uuid.NewString(), waiting.EnvironmentID); err != nil { + t.Fatal(err) + } + owner := runningPressureOwner(t, f, changes, installation, node.NodeID, view.Generation) + + until := time.Now().Add(time.Hour) + + current, err := changes.SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, 5*time.Minute) + if err != nil { + t.Fatal("healthy suspension blocked", err) + } + // Supply healthy native quiesce/capture/stop receipts through the + // existing fenced phase transitions; never settle the stuck owner. + for _, next := range []string{"suspending", "suspended"} { + current, err = changes.SetCompute(t.Context(), current, next, json.RawMessage(`{}`), &until, 5*time.Minute) + if err != nil { + t.Fatal(err) + } + } + + reserved, err := changes.EnsurePlacement(t.Context(), waiting, installation) + if err != nil || reserved.NodeID != node.NodeID { + t.Fatal("waiting input did not advance on healthy node", reserved, err) + } + observed, err := f.adapter.EnvironmentAllocation(t.Context(), stuck.Key()) + if err != nil || observed.ID != stuck.ID || observed.DeviceID != stuck.DeviceID || observed.State != "running" || observed.ComputePhase != phase { + t.Fatal("changed unknown receipt", observed, err) + } + nodes, err := f.adapter.Nodes(t.Context()) + if err != nil { + t.Fatal(err) + } + for _, n := range nodes { + if n.ID == badNode.NodeID && (n.Active != 1 || n.Retained != 1) { + t.Fatal("released unknown capacity", n) + } + } + }) + } +} diff --git a/services/core/internal/persistence/postgres/deploymentpg/tx.go b/services/core/internal/persistence/postgres/deploymentpg/tx.go index 7e865c437..9a2fc2dc0 100644 --- a/services/core/internal/persistence/postgres/deploymentpg/tx.go +++ b/services/core/internal/persistence/postgres/deploymentpg/tx.go @@ -237,7 +237,14 @@ func (t *nodeTx) UpsertGenerationStatus(status deployment.GenerationStatusRecord if err != nil { return err } - return t.q.UpsertNodeGenerationStatus(t.ctx, sqlc.UpsertNodeGenerationStatusParams{NodeID: id, Generation: int64(status.Generation), SpecificationDigest: status.SpecificationDigest, ConnectionID: connection, OwnerEpoch: int64(status.OwnerEpoch), State: status.State, Diagnostic: status.Diagnostic}) + var checkpoint []byte + if status.Checkpoint != nil { + checkpoint, err = json.Marshal(status.Checkpoint) + if err != nil { + return err + } + } + return t.q.UpsertNodeGenerationStatus(t.ctx, sqlc.UpsertNodeGenerationStatusParams{NodeID: id, Generation: int64(status.Generation), SpecificationDigest: status.SpecificationDigest, ConnectionID: connection, OwnerEpoch: int64(status.OwnerEpoch), State: status.State, Diagnostic: status.Diagnostic, Checkpoint: checkpoint}) } func (t *nodeTx) PromoteServingGeneration(nodeID string, generation uint64) error { diff --git a/services/core/internal/persistence/postgres/pgtest/lease.go b/services/core/internal/persistence/postgres/pgtest/lease.go index 551c16f22..8119a50c4 100644 --- a/services/core/internal/persistence/postgres/pgtest/lease.go +++ b/services/core/internal/persistence/postgres/pgtest/lease.go @@ -14,7 +14,8 @@ func ObserveExecutionLeaseRelease(t *testing.T, pool *pgxpool.Pool) func() { t.Helper() // Match the single-bigint key in queries/scheduling.sql, scoped to this DB. const lock = `locktype='advisory' AND granted AND objsubid=1 - AND classid::bigint * 4294967296 + objid::bigint = 706172736172 + AND classid::bigint = (706172736172::bigint >> 32) + AND objid::bigint = (706172736172::bigint & 4294967295) AND database=(SELECT oid FROM pg_database WHERE datname=current_database())` ctx, cancel := context.WithTimeout(t.Context(), 5*time.Second) defer cancel() diff --git a/services/core/internal/persistence/postgres/pgtest/lease_test.go b/services/core/internal/persistence/postgres/pgtest/lease_test.go new file mode 100644 index 000000000..443fb6210 --- /dev/null +++ b/services/core/internal/persistence/postgres/pgtest/lease_test.go @@ -0,0 +1,26 @@ +package pgtest + +import "testing" + +func TestObserveExecutionLeaseReleaseWithNegativeAdvisoryLock(t *testing.T) { + pool := OpenIsolated(t, nil) + conn, err := pool.Acquire(t.Context()) + if err != nil { + t.Fatal(err) + } + defer conn.Release() + // Negative single-bigint keys occupy high unsigned pg_locks.classid bits. + // Other tests may hold one on the same PostgreSQL server while we observe + // this database's positive execution key. + if _, err := conn.Exec(t.Context(), `SELECT pg_advisory_lock(-1::bigint),pg_advisory_lock(706172736172::bigint)`); err != nil { + t.Fatal(err) + } + released := ObserveExecutionLeaseRelease(t, pool) + if _, err := conn.Exec(t.Context(), `SELECT pg_advisory_unlock(706172736172::bigint)`); err != nil { + t.Fatal(err) + } + released() + if _, err := conn.Exec(t.Context(), `SELECT pg_advisory_unlock(-1::bigint)`); err != nil { + t.Fatal(err) + } +} diff --git a/services/core/internal/persistence/postgres/pgunit/lease_test.go b/services/core/internal/persistence/postgres/pgunit/lease_test.go index c5377bc66..d1df04098 100644 --- a/services/core/internal/persistence/postgres/pgunit/lease_test.go +++ b/services/core/internal/persistence/postgres/pgunit/lease_test.go @@ -33,7 +33,8 @@ func awaitLeaseRelease(t *testing.T, pool *pgxpool.Pool) { var held bool // Match the single-bigint key in queries/scheduling.sql, scoped to this database. err := pool.QueryRow(t.Context(), `SELECT EXISTS (SELECT 1 FROM pg_locks WHERE locktype='advisory' AND granted AND objsubid=1 - AND classid::bigint * 4294967296 + objid::bigint = 706172736172 + AND classid::bigint = (706172736172::bigint >> 32) + AND objid::bigint = (706172736172::bigint & 4294967295) AND database=(SELECT oid FROM pg_database WHERE datname=current_database()))`).Scan(&held) if err != nil { t.Fatal("observe execution lease release", err) diff --git a/services/core/internal/persistence/postgres/placementpg/placementpg.go b/services/core/internal/persistence/postgres/placementpg/placementpg.go index 73fc46b35..ad5abde8a 100644 --- a/services/core/internal/persistence/postgres/placementpg/placementpg.go +++ b/services/core/internal/persistence/postgres/placementpg/placementpg.go @@ -7,6 +7,7 @@ package placementpg import ( "context" + "encoding/json" "github.com/google/uuid" "github.com/jackc/pgx/v5/pgtype" @@ -14,6 +15,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/db/sqlc" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/pgunit" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" ) // LockDeployment locks the deployment and returns it as placement reads it. @@ -65,21 +67,34 @@ func ReleasePlacement(ctx context.Context, q *sqlc.Queries, environment pgtype.U return q.ReleaseRuntimePlacement(ctx, environment) } -// LoadRestore locks the deployment and returns what restoring suspended -// compute of the generation on the node reads. -func LoadRestore(ctx context.Context, q *sqlc.Queries, nodeID pgtype.UUID, generation uint64) (placement.Restore, error) { - if _, err := q.LockRuntimeDeployment(ctx); err != nil { - return placement.Restore{}, err +// LoadCheckpointNodes joins current presence/capacity with exact-generation +// readiness. The caller holds the deployment lock for both observations. +func LoadCheckpointNodes(ctx context.Context, q *sqlc.Queries, generation uint64, nodes []placement.Node) ([]placement.CheckpointNode, error) { + ready, err := q.ListCheckpointGenerationNodes(ctx, int64(generation)) + if err != nil { + return nil, err + } + byID := make(map[string]placement.Node, len(nodes)) + for _, node := range nodes { + byID[node.ID] = node } - restore := placement.Restore{Generation: generation} - rows, err := q.ListRuntimeNodes(ctx, nodeID) - if err != nil || len(rows) == 0 { - return restore, err + candidates := make([]placement.CheckpointNode, 0, len(ready)) + for _, row := range ready { + if len(row.Checkpoint) == 0 { + continue + } + var compatibility sandbox.CheckpointCompatibility + if err := json.Unmarshal(row.Checkpoint, &compatibility); err != nil { + return nil, err + } + if err := compatibility.Validate(); err != nil { + return nil, err + } + if node, ok := byID[uuidString(row.ID)]; ok { + candidates = append(candidates, placement.CheckpointNode{Node: node, Checkpoint: &compatibility}) + } } - n := node(rows[0]) - restore.Node = &n - restore.GenerationReady, err = q.NodeGenerationReady(ctx, sqlc.NodeGenerationReadyParams{NodeID: nodeID, Generation: int64(generation)}) - return restore, err + return candidates, nil } // ComputeBlocksAdmission reports whether the Session's managed compute is in diff --git a/services/core/internal/persistence/postgres/sessionpg/artifacts.go b/services/core/internal/persistence/postgres/sessionpg/artifacts.go index 21401d11d..a43074fee 100644 --- a/services/core/internal/persistence/postgres/sessionpg/artifacts.go +++ b/services/core/internal/persistence/postgres/sessionpg/artifacts.go @@ -173,7 +173,7 @@ type artifactStaging struct { func (t *artifactStaging) LoadEnvironment(ctx context.Context) (sessions.Environment, error) { row, err := t.q.GetSessionEnvironment(ctx, sqlc.GetSessionEnvironmentParams{TenantID: t.turn.TenantID, ID: t.turn.SessionID}) - return environmentFromRow(row.Environment, row.TenantID, row.Configuration, err) + return environmentFromRow(row.Environment, row.TenantID, row.Configuration, row.ExternalWorkspace, err) } func (t *artifactStaging) PutArtifactContent(ctx context.Context, path string, size int64, content io.Reader) error { diff --git a/services/core/internal/persistence/postgres/sessionpg/binding.go b/services/core/internal/persistence/postgres/sessionpg/binding.go index dad3703c5..6650c2552 100644 --- a/services/core/internal/persistence/postgres/sessionpg/binding.go +++ b/services/core/internal/persistence/postgres/sessionpg/binding.go @@ -116,7 +116,7 @@ func (t *SessionTx) LoadEnvironmentInput(ctx context.Context) (*sessions.Environ func (t *SessionTx) LoadEnvironment(ctx context.Context) (sessions.Environment, error) { row, err := t.q.GetSessionEnvironment(ctx, sqlc.GetSessionEnvironmentParams{TenantID: t.tenant, ID: t.session}) - return environmentFromRow(row.Environment, row.TenantID, row.Configuration, err) + return environmentFromRow(row.Environment, row.TenantID, row.Configuration, row.ExternalWorkspace, err) } // RecordEnvironmentFailure fails the Environment only while it is the @@ -185,7 +185,10 @@ func (t *SessionTx) InsertEnvironmentDevice(ctx context.Context, device sessions if err != nil { return err } - _, err = t.q.BindSessionDevice(ctx, sqlc.BindSessionDeviceParams{TenantID: t.tenant, ID: t.session, ID_2: inserted}) + _, err = t.q.BindHostedSessionDevice(ctx, sqlc.BindHostedSessionDeviceParams{TenantID: t.tenant, ID: t.session, ID_2: inserted}) + if errors.Is(err, pgx.ErrNoRows) { + return sessions.ErrDeviceBindingConflict + } return err } diff --git a/services/core/internal/persistence/postgres/sessionpg/creation.go b/services/core/internal/persistence/postgres/sessionpg/creation.go index 7861947d8..127fe6752 100644 --- a/services/core/internal/persistence/postgres/sessionpg/creation.go +++ b/services/core/internal/persistence/postgres/sessionpg/creation.go @@ -102,14 +102,6 @@ func (t *creationTx) LockDeployment(ctx context.Context) (placement.Deployment, return placementpg.LockDeployment(ctx, t.q) } -func (t *creationTx) LoadNodes(ctx context.Context) ([]placement.Node, error) { - return placementpg.LoadNodes(ctx, t.q) -} - -func (t *creationTx) ReservePlacement(ctx context.Context, chosen placement.Placement) error { - return placementpg.ReservePlacement(ctx, t.q, t.session, chosen) -} - func (t *creationTx) LockSkills(ctx context.Context, ids []string) (map[string]skills.Skill, error) { locked, err := skillpg.LockSkills(ctx, t.q, t.tenant, ids) return locked, skillError(err) diff --git a/services/core/internal/persistence/postgres/sessionpg/creation_test.go b/services/core/internal/persistence/postgres/sessionpg/creation_test.go index 2165b2c07..bc2b5326b 100644 --- a/services/core/internal/persistence/postgres/sessionpg/creation_test.go +++ b/services/core/internal/persistence/postgres/sessionpg/creation_test.go @@ -464,10 +464,9 @@ func TestEnvironmentCreationReservesItsInitialInput(t *testing.T) { } // Hosted creation checks admission on the locked deployment, so it sees a -// reset committed while it waited, and places after creating the -// Environment, rolling both back when no node is available. A retry admits -// nothing and still returns its Session. -func TestHostedCreationAdmitsAndPlacesUnderTheDeploymentLock(t *testing.T) { +// reset committed while it waited. Accepted Environments wait for compute +// without a placement; a retry admits nothing and returns its Session. +func TestHostedCreationAdmitsUnderDeploymentLockAndLeavesComputeUnreserved(t *testing.T) { pool := pgtest.OpenIsolated(t, nil) _, service := creationService(t, pool) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) @@ -518,11 +517,18 @@ func TestHostedCreationAdmitsAndPlacesUnderTheDeploymentLock(t *testing.T) { t.Fatal("retry ran admission", retry, err) } exec(t, pool, "UPDATE runtime_deployment SET reset_clear=NULL, reset_requested_at=NULL, reset_forced_at=NULL, reset_audit=NULL") - if _, err := service.CreateSession(ctx, tenant, hosted("unplaced")); !errors.Is(err, placement.ErrNodeUnavailable) { - t.Fatal("placement without a node", err) + queued, err := service.CreateSession(ctx, tenant, hosted("unplaced")) + if err != nil || !queued.Created { + t.Fatal("creation without available compute was not accepted", queued, err) } - if count := countRows(t, pool, "SELECT count(*) FROM sessions s JOIN environments e ON e.session_id=s.id WHERE s.tenant_id=$1", tenant); count != 1 { - t.Fatal("failed hosted creation left work", count) + if count := countRows(t, pool, "SELECT count(*) FROM sessions s JOIN environments e ON e.session_id=s.id WHERE s.tenant_id=$1", tenant); count != 2 { + t.Fatal("accepted hosted creation did not preserve both Sessions", count) + } + if count := countRows(t, pool, "SELECT count(*) FROM runtime_placements WHERE environment_id=$1", queued.Session.Environment.ID); count != 0 { + t.Fatal("creation reserved a node", count) + } + if count := countRows(t, pool, "SELECT count(*) FROM runtime_allocations WHERE environment_id=$1", queued.Session.Environment.ID); count != 0 { + t.Fatal("creation allocated compute", count) } } @@ -644,3 +650,65 @@ func TestCreationAudit(t *testing.T) { t.Fatal("created resources", owned, rows.Err()) } } + +// Profile admission uses the deployment selected while holding its lock, so a +// storage-mode update cannot race a preflight check and persist unsupported work. +func TestHostedExternalHistoryAdmissionRollsBackAfterDeploymentChange(t *testing.T) { + pool := pgtest.OpenIsolated(t, nil) + store, _ := creationService(t, pool) + rules, err := placement.NewRules(externalAdmissionDeclarations{}, "https://core.example") + if err != nil { + t.Fatal(err) + } + service, err := sessions.NewService(store, rules) + if err != nil { + t.Fatal(err) + } + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + tenant := uuid.NewString() + exec(t, pool, `UPDATE runtime_deployment SET installation_id=$1, backend_fingerprint=$2, provider_kind='test-direct', mode='direct', generation=1, specification='{"resources":{"cpus":2,"memory_mib":2048}}'`, uuid.New(), strings.Repeat("a", 64)) + input := sessions.CreateSession{Creator: creator, Engine: "codex", IdempotencyKey: "raced", Configuration: json.RawMessage(`{"agent":{"model":"m"},"environment":{"type":"openai_hosted"}}`)} + tx, err := pool.Begin(ctx) + if err != nil { + t.Fatal(err) + } + defer tx.Rollback(context.Background()) + var holder int32 + if err = tx.QueryRow(ctx, "SELECT pg_backend_pid() FROM runtime_deployment FOR UPDATE").Scan(&holder); err != nil { + t.Fatal(err) + } + done := make(chan error, 1) + go func() { _, err := service.CreateSession(ctx, tenant, input); done <- err }() + awaitBlocked(ctx, t, pool, holder) + if _, err = tx.Exec(ctx, `UPDATE runtime_deployment SET specification=jsonb_set(specification,'{workspace}','{}')`); err != nil { + t.Fatal(err) + } + if err = tx.Commit(ctx); err != nil { + t.Fatal(err) + } + if err = <-done; !errors.Is(err, sessions.ErrInvalidInput) { + t.Fatal("external combination admitted", err) + } + if count := countRows(t, pool, "SELECT count(*) FROM sessions WHERE tenant_id=$1", tenant); count != 0 { + t.Fatal("rejected admission persisted Session", count) + } + input.SupportsRetainedNativeHistory = true + created, err := service.CreateSession(ctx, tenant, input) + if err != nil { + t.Fatal(err) + } + input.SupportsRetainedNativeHistory = false + retry, err := service.CreateSession(ctx, tenant, input) + if err != nil || retry.Created || retry.Session.ID != created.Session.ID { + t.Fatal("trusted admission input changed retry identity", retry, err) + } +} + +// This fixture declares a direct provider supporting both storage modes. +type externalAdmissionDeclarations struct{} + +func (externalAdmissionDeclarations) RequiresPublicOrigin(string) (bool, error) { return false, nil } +func (externalAdmissionDeclarations) ValidateSpecification(string, sandbox.DeploymentSpec) error { + return nil +} diff --git a/services/core/internal/persistence/postgres/sessionpg/device_capabilities_test.go b/services/core/internal/persistence/postgres/sessionpg/device_capabilities_test.go new file mode 100644 index 000000000..c468bc51b --- /dev/null +++ b/services/core/internal/persistence/postgres/sessionpg/device_capabilities_test.go @@ -0,0 +1,46 @@ +package sessionpg + +import ( + "strings" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/pgtest" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/pgunit" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/runtimedevice" + "github.com/jackc/pgx/v5/pgtype" +) + +func TestAuthenticatedCapabilitiesKeepRevokedReceipt(t *testing.T) { + pool := pgtest.OpenIsolated(t, nil) + tenant, _, _ := newEnvironment(t, pool, "self_hosted", "pending") + device := newDevice(t, pool, tenant, pgtype.UUID{}) + store := New(pgunit.NewPool(pool), nil) + kinds := []runtimedevice.SupportedAgentKind{{Kind: "fixture", Available: true, Capabilities: runtimedevice.KindCapabilities{RetainedNativeHistory: true}}} + correct := strings.Repeat("a", 64) + if ok, err := store.TouchAuthenticatedDevice(t.Context(), device, correct, kinds); err != nil || !ok { + t.Fatal(ok, err) + } + read := func() string { + t.Helper() + var raw string + if err := pool.QueryRow(t.Context(), `SELECT supported_agent_kinds::text FROM devices WHERE id=$1`, device).Scan(&raw); err != nil { + t.Fatal(err) + } + return raw + } + before := read() + if ok, err := store.TouchAuthenticatedDevice(t.Context(), device, strings.Repeat("b", 64), nil); err != nil || ok || read() != before { + t.Fatal("wrong credential changed receipt", ok, err) + } + if ok, err := store.TouchAuthenticatedDevice(t.Context(), device, correct, nil); err != nil || !ok || read() != "[]" { + t.Fatal("unknown capabilities not cleared", ok, err) + } + if _, err := store.TouchAuthenticatedDevice(t.Context(), device, correct, kinds); err != nil { + t.Fatal(err) + } + before = read() + exec(t, pool, `UPDATE devices SET revoked_at=clock_timestamp() WHERE id=$1`, device) + if ok, err := store.TouchAuthenticatedDevice(t.Context(), device, correct, nil); err != nil || ok || read() != before { + t.Fatal("revoked device changed retained receipt", ok, err) + } +} diff --git a/services/core/internal/persistence/postgres/sessionpg/devices.go b/services/core/internal/persistence/postgres/sessionpg/devices.go index 7512586b3..188a45ef0 100644 --- a/services/core/internal/persistence/postgres/sessionpg/devices.go +++ b/services/core/internal/persistence/postgres/sessionpg/devices.go @@ -2,6 +2,7 @@ package sessionpg import ( "context" + "encoding/json" "errors" "fmt" @@ -196,12 +197,19 @@ func (s *Store) TouchDevice(ctx context.Context, device string) (bool, error) { return n > 0, err } -func (s *Store) TouchAuthenticatedDevice(ctx context.Context, device, credentialHash string) (bool, error) { +func (s *Store) TouchAuthenticatedDevice(ctx context.Context, device, credentialHash string, kinds []runtimedevice.SupportedAgentKind) (bool, error) { id, err := parseID(device) if err != nil { return false, err } - n, err := s.units.Queries().TouchAuthenticatedDevice(ctx, sqlc.TouchAuthenticatedDeviceParams{ID: id, CredentialHash: credentialHash}) + if kinds == nil { + kinds = []runtimedevice.SupportedAgentKind{} + } + raw, err := json.Marshal(kinds) + if err != nil { + return false, err + } + n, err := s.units.Queries().TouchAuthenticatedDevice(ctx, sqlc.TouchAuthenticatedDeviceParams{ID: id, CredentialHash: credentialHash, SupportedAgentKinds: raw}) return n > 0, err } diff --git a/services/core/internal/persistence/postgres/sessionpg/environment.go b/services/core/internal/persistence/postgres/sessionpg/environment.go index 830d095ab..e191fe228 100644 --- a/services/core/internal/persistence/postgres/sessionpg/environment.go +++ b/services/core/internal/persistence/postgres/sessionpg/environment.go @@ -23,7 +23,7 @@ func LoadEnvironment(ctx context.Context, q *sqlc.Queries, tenant, environment s return sessions.Environment{}, err } row, err := q.GetEnvironment(ctx, sqlc.GetEnvironmentParams{TenantID: tenantID, ID: pgunit.PathID(environment)}) - return environmentFromRow(row.Environment, row.TenantID, row.Configuration, err) + return environmentFromRow(row.Environment, row.TenantID, row.Configuration, row.ExternalWorkspace, err) } // loadSessionEnvironment reads, on q, the Environment of the tenant's Session @@ -39,7 +39,7 @@ func loadSessionEnvironment(ctx context.Context, q *sqlc.Queries, tenant, sessio return sessions.Environment{}, err } row, err := q.GetSessionEnvironment(ctx, sqlc.GetSessionEnvironmentParams{TenantID: tenantID, ID: id}) - return environmentFromRow(row.Environment, row.TenantID, row.Configuration, err) + return environmentFromRow(row.Environment, row.TenantID, row.Configuration, row.ExternalWorkspace, err) } func (s *Store) GetEnvironment(ctx context.Context, tenant, environment string) (sessions.Environment, error) { diff --git a/services/core/internal/persistence/postgres/sessionpg/execution_inputs_test.go b/services/core/internal/persistence/postgres/sessionpg/execution_inputs_test.go index 80af513a5..0cedc377b 100644 --- a/services/core/internal/persistence/postgres/sessionpg/execution_inputs_test.go +++ b/services/core/internal/persistence/postgres/sessionpg/execution_inputs_test.go @@ -34,7 +34,8 @@ func terminateLeaseOwner(t *testing.T, pool *pgxpool.Pool) { t.Helper() var killed bool err := pool.QueryRow(t.Context(), `SELECT pg_terminate_backend(pid, 1000) FROM pg_locks WHERE locktype='advisory' AND granted AND objsubid=1 - AND classid::bigint * 4294967296 + objid::bigint = 706172736172 + AND classid::bigint = (706172736172::bigint >> 32) + AND objid::bigint = (706172736172::bigint & 4294967295) AND database=(SELECT oid FROM pg_database WHERE datname=current_database())`).Scan(&killed) if err != nil || !killed { t.Fatal("terminate execution lease owner", killed, err) diff --git a/services/core/internal/persistence/postgres/sessionpg/rows.go b/services/core/internal/persistence/postgres/sessionpg/rows.go index d635b72bf..4cbe2ddf0 100644 --- a/services/core/internal/persistence/postgres/sessionpg/rows.go +++ b/services/core/internal/persistence/postgres/sessionpg/rows.go @@ -101,7 +101,7 @@ func sessionCreator(kind, id pgtype.Text) (*identity.Subject, error) { // environmentFromRow maps a stored Environment of the tenant with its // normalized configuration snapshot, from the result of the query that read // it: no rows is sessions.ErrNotFound. -func environmentFromRow(row sqlc.Environment, tenant pgtype.UUID, configuration []byte, err error) (sessions.Environment, error) { +func environmentFromRow(row sqlc.Environment, tenant pgtype.UUID, configuration []byte, externalWorkspace bool, err error) (sessions.Environment, error) { if errors.Is(err, pgx.ErrNoRows) { return sessions.Environment{}, sessions.ErrNotFound } @@ -115,6 +115,6 @@ func environmentFromRow(row sqlc.Environment, tenant pgtype.UUID, configuration return sessions.Environment{ ID: uuid.UUID(row.ID.Bytes).String(), SessionID: uuid.UUID(row.SessionID.Bytes).String(), TenantID: uuid.UUID(tenant.Bytes).String(), Status: row.Status, - Initialization: row.Initialization, CreatedAt: row.CreatedAt.Time, Configuration: configuration, + ExternalWorkspace: externalWorkspace, Initialization: row.Initialization, CreatedAt: row.CreatedAt.Time, Configuration: configuration, }, nil } diff --git a/services/core/internal/persistence/postgres/sessionpg/session.go b/services/core/internal/persistence/postgres/sessionpg/session.go index 2daebf41a..222db15d4 100644 --- a/services/core/internal/persistence/postgres/sessionpg/session.go +++ b/services/core/internal/persistence/postgres/sessionpg/session.go @@ -24,7 +24,7 @@ func LockSession(ctx context.Context, q *sqlc.Queries, tenant, session pgtype.UU if err != nil { return sessions.LockedSession{}, err } - return sessions.LockedSession{Deleted: row.DeletedAt.Valid}, nil + return sessions.LockedSession{Deleted: row.DeletedAt.Valid, Engine: row.Engine}, nil } // WithSession runs one Session transaction on runner: it locks the tenant's diff --git a/services/core/internal/persistence/postgres/sessionpg/session_reads.go b/services/core/internal/persistence/postgres/sessionpg/session_reads.go index c987a5fdd..e38c2bd20 100644 --- a/services/core/internal/persistence/postgres/sessionpg/session_reads.go +++ b/services/core/internal/persistence/postgres/sessionpg/session_reads.go @@ -285,7 +285,7 @@ func loadSessionActivity(ctx context.Context, q *sqlc.Queries, session sessions. tenant, _ := parseID(session.TenantID) environment, err := q.GetSessionEnvironment(ctx, sqlc.GetSessionEnvironmentParams{TenantID: tenant, ID: id}) if err == nil { - value, err := environmentFromRow(environment.Environment, environment.TenantID, environment.Configuration, nil) + value, err := environmentFromRow(environment.Environment, environment.TenantID, environment.Configuration, environment.ExternalWorkspace, nil) if err != nil { return session, err } diff --git a/services/core/internal/runtimedevice/state.go b/services/core/internal/runtimedevice/state.go index 73f917dc1..fd27fc78d 100644 --- a/services/core/internal/runtimedevice/state.go +++ b/services/core/internal/runtimedevice/state.go @@ -65,6 +65,7 @@ type KindCapabilities struct { Usage bool `json:"usage,omitempty"` Resume bool `json:"resume,omitempty"` NativeSessionRecovery bool `json:"native_session_recovery,omitempty"` + RetainedNativeHistory bool `json:"retained_native_history,omitempty"` Steering bool `json:"steering,omitempty"` MessageItems bool `json:"message_items,omitempty"` diff --git a/services/core/internal/runtimegateway/runtime_prepare.go b/services/core/internal/runtimegateway/runtime_prepare.go index 77e6493f9..677928f6a 100644 --- a/services/core/internal/runtimegateway/runtime_prepare.go +++ b/services/core/internal/runtimegateway/runtime_prepare.go @@ -50,8 +50,10 @@ func (s *Session) PrepareRuntime(ctx context.Context, id string, request proto.R s.capabilities = map[string]chan proto.Envelope{id: replies} s.capabilitiesMu.Unlock() defer func() { s.capabilitiesMu.Lock(); delete(s.capabilities, id); s.capabilitiesMu.Unlock() }() - ctx, cancel := context.WithTimeout(ctx, 195*time.Second) + ctx, cancel := context.WithTimeout(ctx, time.Duration(request.BudgetMS)*time.Millisecond) defer cancel() + transfer, stopTransfer := context.WithTimeout(ctx, time.Duration(proto.RuntimePrepareTransferBudgetMS)*time.Millisecond) + defer stopTransfer() exchange := func(payload proto.RuntimePreparePayload, outcome string, offset int) (proto.RuntimePrepareResultPayload, error) { env, err := proto.NewEnvelope(proto.TypeRuntimePrepare, id, payload) if err != nil { @@ -61,7 +63,13 @@ func (s *Session) PrepareRuntime(ctx context.Context, id string, request proto.R if err != nil || len(encoded) > proto.RuntimePrepareMaxFrameBytes { return unknown, errors.New("agentdaemon gateway: invalid Runtime frame") } - reply, err := s.exchangeChunkFrame(ctx, env, replies) + operation := ctx + if outcome != "completed" { + operation = transfer + } else if err := transfer.Err(); err != nil { + return unknown, err + } + reply, err := s.exchangeChunkFrame(operation, env, replies) if err != nil { return unknown, err } diff --git a/services/core/internal/runtimegateway/runtime_prepare_test.go b/services/core/internal/runtimegateway/runtime_prepare_test.go index 02834d87a..f0beea586 100644 --- a/services/core/internal/runtimegateway/runtime_prepare_test.go +++ b/services/core/internal/runtimegateway/runtime_prepare_test.go @@ -9,6 +9,7 @@ import ( "errors" "fmt" "testing" + "testing/synctest" "time" "github.com/MiniMax-AI/OpenAgentCore/internal/agentcapabilities" @@ -18,7 +19,7 @@ import ( ) func skillPreparation() proto.RuntimePreparePayload { - return proto.RuntimePreparePayload{EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "skill", + return proto.RuntimePreparePayload{BudgetMS: 300000, EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "skill", Skill: &agentskill.Metadata{Type: "inline", Name: "example", Description: "Example"}} } @@ -116,7 +117,7 @@ func TestCapabilitiesTransfersMoreThanFrameLimitAndCorrelates(t *testing.T) { func TestCapabilitiesFinalizeTransfersNoArchive(t *testing.T) { s := NewSession(newFakeConn(), "device", "tenant", "test", nil, nil) defer s.Close("test") - request := proto.RuntimePreparePayload{EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "finalize", Sources: &agentcapabilities.Input{}} + request := proto.RuntimePreparePayload{BudgetMS: 300000, EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "finalize", Sources: &agentcapabilities.Input{}} id := uuid.NewString() done := beginCapabilities(s, t.Context(), id, request, nil) env := nextCapabilityFrame(t, s) @@ -239,7 +240,7 @@ func TestRuntimeInitialFileChunking(t *testing.T) { t.Run(fmt.Sprint(size), func(t *testing.T) { s := NewSession(newFakeConn(), "device", "tenant", "test", nil, nil) defer s.Close("test") - request := proto.RuntimePreparePayload{EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "file", File: &proto.RuntimeInitialFile{Path: "/workspace/project/file"}} + request := proto.RuntimePreparePayload{BudgetMS: 300000, EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "file", File: &proto.RuntimeInitialFile{Path: "/workspace/project/file"}} data := bytes.Repeat([]byte("z"), size) id := uuid.NewString() done := beginCapabilities(s, t.Context(), id, request, data) @@ -289,7 +290,7 @@ func TestRuntimeInitializationNoDataAndExitReceipt(t *testing.T) { t.Run(fmt.Sprint(exit), func(t *testing.T) { s := NewSession(newFakeConn(), "device", "tenant", "test", nil, nil) defer s.Close("test") - request := proto.RuntimePreparePayload{EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "initialize", Initialization: &proto.RuntimeInitialization{Action: "setup", Command: "echo test"}} + request := proto.RuntimePreparePayload{BudgetMS: 300000, EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "initialize", Initialization: &proto.RuntimeInitialization{Action: "setup", Command: "echo test"}} if _, err := s.PrepareRuntime(t.Context(), uuid.NewString(), request, []byte("forbidden")); err == nil { t.Fatal("initialization body accepted") } @@ -327,3 +328,72 @@ func TestRuntimeInitializationNoDataAndExitReceipt(t *testing.T) { }) } } + +func TestRuntimePreparationSeparatesTransferAndApplyBudgets(t *testing.T) { + for _, commit := range []bool{false, true} { + t.Run(fmt.Sprint(commit), func(t *testing.T) { + synctest.Test(t, func(t *testing.T) { + s := NewSession(newFakeConn(), "device", "tenant", "test", nil, nil) + defer s.Close("test") + request := proto.RuntimePreparePayload{BudgetMS: 300000, EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "initialize", Initialization: &proto.RuntimeInitialization{Action: "configure"}} + id := uuid.NewString() + done := beginCapabilities(s, t.Context(), id, request, nil) + frame := nextCapabilityFrame(t, s) + var sent proto.RuntimePreparePayload + if frame.DecodePayload(&sent) != nil || sent.BudgetMS != request.BudgetMS { + t.Fatal("begin budget lost") + } + if commit { + replyCapabilities(s, id, proto.RuntimePrepareResultPayload{Outcome: "ready"}) + nextCapabilityFrame(t, s) + } + time.Sleep(121 * time.Second) + synctest.Wait() + if !commit { + result := finishCapabilities(t, done) + if !errors.Is(result.err, context.DeadlineExceeded) || result.result.Outcome != "unknown" { + t.Fatal(result) + } + return + } + select { + case result := <-done: + t.Fatal("transfer timer clamped execution", result) + default: + } + time.Sleep(180 * time.Second) + result := finishCapabilities(t, done) + if !errors.Is(result.err, context.DeadlineExceeded) || result.result.Outcome != "unknown" { + t.Fatal(result) + } + }) + }) + } +} + +func TestRuntimePreparationCommittedApplyStopsWaitingOnDisconnectOrCallerDeadline(t *testing.T) { + for _, disconnect := range []bool{false, true} { + t.Run(fmt.Sprint(disconnect), func(t *testing.T) { + s := NewSession(newFakeConn(), "device", "tenant", "test", nil, nil) + defer s.Close("test") + ctx, cancel := context.WithTimeout(t.Context(), 100*time.Millisecond) + defer cancel() + request := proto.RuntimePreparePayload{BudgetMS: 300000, EnvironmentID: uuid.NewString(), SessionID: uuid.NewString(), Action: "initialize", Initialization: &proto.RuntimeInitialization{Action: "configure"}} + id := uuid.NewString() + done := beginCapabilities(s, ctx, id, request, nil) + nextCapabilityFrame(t, s) + replyCapabilities(s, id, proto.RuntimePrepareResultPayload{Outcome: "ready"}) + nextCapabilityFrame(t, s) + expected := error(context.DeadlineExceeded) + if disconnect { + s.Close("disconnect during apply") + expected = ErrSessionClosed + } + result := finishCapabilities(t, done) + if !errors.Is(result.err, expected) || result.result.Outcome != "unknown" { + t.Fatal(result) + } + noCapabilityFrame(t, s) + }) + } +} diff --git a/services/core/internal/runtimegateway/session.go b/services/core/internal/runtimegateway/session.go index bb1f5acd5..a3b7adcfe 100644 --- a/services/core/internal/runtimegateway/session.go +++ b/services/core/internal/runtimegateway/session.go @@ -518,6 +518,7 @@ func deviceKindsFromHeartbeat(p proto.HeartbeatPayload) []runtimedevice.Supporte DurableTurns: info.Capabilities.DurableTurns.IsSupported(), DurableInputReceipts: info.Capabilities.DurableInputReceipts.IsSupported(), NativeSessionRecovery: info.Capabilities.NativeSessionRecovery.IsSupported(), + RetainedNativeHistory: info.Capabilities.RetainedNativeHistory.IsSupported(), MessageItems: info.Capabilities.MessageItems.IsSupported(), ToolObservations: info.Capabilities.ToolObservations.IsSupported(), diff --git a/services/core/internal/sandbox/docker/node.go b/services/core/internal/sandbox/docker/node.go index 115d6bf0c..b8ddc04fb 100644 --- a/services/core/internal/sandbox/docker/node.go +++ b/services/core/internal/sandbox/docker/node.go @@ -68,7 +68,9 @@ func BuildNode(config sandbox.NodeConfig, _ sandbox.LocalOptions, result *sandbo return func() {}, errors.New("invalid managed Docker provider configuration") } result.Provider = provider - result.Probe = dockerProbe(c, entry.Image, config.Specification.Resources) + result.Probe = func(ctx context.Context) (*sandbox.CheckpointCompatibility, error) { + return nil, dockerProbe(c, entry.Image, config.Specification.Resources)(ctx) + } result.BackendFingerprint = sandbox.BackendFingerprint(config.Provider, entry.Host) return closeProvider, nil } diff --git a/services/core/internal/sandbox/e2b/suspension.go b/services/core/internal/sandbox/e2b/suspension.go index 5237ea082..0b55260e0 100644 --- a/services/core/internal/sandbox/e2b/suspension.go +++ b/services/core/internal/sandbox/e2b/suspension.go @@ -81,6 +81,12 @@ func (p *Provider) Resume(ctx context.Context, q sandbox.ResumeRequest) (sandbox if err != nil { return sandbox.ComputeState{}, err } + if out.State != nil && out.State.RestoreAttemptClosed != "" { + if !out.State.ClosesRestoreAttempt(q) { + return sandbox.ComputeState{}, sandbox.ErrOwnership + } + return *out.State, nil + } if out.State == nil || sandbox.ValidateComputeResult(q.Target, out.State.Compute) != nil || out.State.Status != "running" || !out.State.BootstrapComplete { return sandbox.ComputeState{}, sandbox.ErrComputeUnconfirmed } diff --git a/services/core/internal/sandbox/e2b/suspension_test.go b/services/core/internal/sandbox/e2b/suspension_test.go index d4a0c0c06..5955c2223 100644 --- a/services/core/internal/sandbox/e2b/suspension_test.go +++ b/services/core/internal/sandbox/e2b/suspension_test.go @@ -46,3 +46,22 @@ func TestSuspensionRejectsUnsettledAndForeignHelperOutcomes(t *testing.T) { t.Fatal(err) } } + +func TestResumeClosedAttemptPreservesInPlaceNativeIdentity(t *testing.T) { + p, f, r := fixture(t) + retained := sandbox.RetainedState{Reference: r.AllocationID, ID: uuid.NewString(), OperationID: uuid.NewString(), SourceName: r.AllocationID, SourceID: "native-id", Data: "opaque"} + target, err := p.NewCompute(bounded(t), r, 1, &retained) + if err != nil { + t.Fatal(err) + } + q := sandbox.ResumeRequest{Reference: r, OperationID: uuid.NewString(), Retained: retained, Target: target, ReconcileOnly: true} + f.response.State = &sandbox.ComputeState{Compute: target, Status: "absent", RestoreAttemptClosed: q.OperationID} + got, err := p.Resume(bounded(t), q) + if err != nil || !got.ClosesRestoreAttempt(q) { + t.Fatal("in-place closure lost", got, err) + } + f.response.State.Compute.ID = "foreign-id" + if _, err := p.Resume(bounded(t), q); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal("foreign native closure accepted", err) + } +} diff --git a/services/core/internal/sandbox/generation.go b/services/core/internal/sandbox/generation.go index f191d2859..66bbbda8d 100644 --- a/services/core/internal/sandbox/generation.go +++ b/services/core/internal/sandbox/generation.go @@ -10,10 +10,11 @@ import ( // GenerationStatus is one sparse observation, not a complete provider inventory. // A serving pin is owned by Core and is independent of target preparation. type GenerationStatus struct { - Generation uint64 `json:"generation"` - SpecificationDigest string `json:"specification_digest"` - State string `json:"state"` - Diagnostic string `json:"diagnostic,omitempty"` + Generation uint64 `json:"generation"` + SpecificationDigest string `json:"specification_digest"` + State string `json:"state"` + Diagnostic string `json:"diagnostic,omitempty"` + Checkpoint *CheckpointCompatibility `json:"checkpoint,omitempty"` } type GenerationReference struct { diff --git a/services/core/internal/sandbox/microsandbox/checkpoint_linux.go b/services/core/internal/sandbox/microsandbox/checkpoint_linux.go new file mode 100644 index 000000000..e28018709 --- /dev/null +++ b/services/core/internal/sandbox/microsandbox/checkpoint_linux.go @@ -0,0 +1,133 @@ +//go:build linux + +package microsandbox + +import ( + "crypto/sha256" + "encoding/hex" + "encoding/json" + "io" + "os" + "path/filepath" + "sort" + "strings" + "syscall" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +// CheckpointStore opens only an explicitly configured, private archive root. +// Its marker is an operator-owned identity, never inferred from workspace paths. +func CheckpointStore(c Config) (string, error) { + if !filepath.IsAbs(c.CheckpointRoot) || filepath.Clean(c.CheckpointRoot) != c.CheckpointRoot { + return "", sandbox.ErrInvalid + } + real, err := filepath.EvalSymlinks(c.CheckpointRoot) + if err != nil { + return "", err + } + if real != c.CheckpointRoot { + return "", sandbox.ErrOwnership + } + fd, err := syscall.Open(c.CheckpointRoot, syscall.O_RDONLY|syscall.O_DIRECTORY|syscall.O_NOFOLLOW|syscall.O_CLOEXEC, 0) + if err != nil { + return "", err + } + f := os.NewFile(uintptr(fd), c.CheckpointRoot) + defer f.Close() + st, err := f.Stat() + if err != nil { + return "", err + } + if st.Mode().Perm() != 0700 || st.Sys().(*syscall.Stat_t).Uid != uint32(os.Geteuid()) { + return "", sandbox.ErrOwnership + } + marker, err := syscall.Openat(fd, ".oac-checkpoint-store", syscall.O_RDONLY|syscall.O_NOFOLLOW|syscall.O_CLOEXEC, 0) + if err != nil { + return "", err + } + m := os.NewFile(uintptr(marker), "checkpoint marker") + defer m.Close() + st, err = m.Stat() + if err != nil { + return "", err + } + if !st.Mode().IsRegular() || st.Mode().Perm() != 0600 || st.Sys().(*syscall.Stat_t).Uid != uint32(os.Geteuid()) { + return "", sandbox.ErrOwnership + } + data, err := io.ReadAll(io.LimitReader(m, 38)) + if err != nil { + return "", err + } + id := strings.TrimSuffix(string(data), "\n") + if !validID(id) || string(data) != id+"\n" { + return "", sandbox.ErrOwnership + } + h := sha256.Sum256([]byte(c.InstallationID + ":" + id)) + return hex.EncodeToString(h[:]), nil +} + +// CheckpointClass deliberately admits only identical native and host execution +// profiles. Paths, hostname and local SDK cache identities are not compatibility. +func CheckpointClass(c Config) (sandbox.CheckpointCompatibility, error) { + domain, err := CheckpointStore(c) + if err != nil { + return sandbox.CheckpointCompatibility{}, err + } + cpu, err := os.ReadFile("/proc/cpuinfo") + if err != nil { + return sandbox.CheckpointCompatibility{}, err + } + stable := checkpointCPUProfiles(string(cpu)) + if len(stable) == 0 { + return sandbox.CheckpointCompatibility{}, sandbox.ErrInvalid + } + kernel, err := os.ReadFile("/proc/sys/kernel/osrelease") + if err != nil { + return sandbox.CheckpointCompatibility{}, err + } + raw, _ := json.Marshal(struct { + SDK, Runtime, Firmware, Image, Kernel string + CPUs uint8 + Memory, Root, Environment uint32 + CPU []string + }{SDKVersion, c.RuntimeSHA256, c.FirmwareSHA256, c.Image, strings.TrimSpace(string(kernel)), c.CPUs, c.MemoryMiB, c.RootDiskMiB, c.EnvironmentDiskMiB, stable}) + h := sha256.Sum256(raw) + return sandbox.CheckpointCompatibility{ArtifactDomain: domain, ExecutionClass: hex.EncodeToString(h[:])}, nil +} + +// Preserve each processor's feature/model association. Multiplicity and logical +// processor indexes are irrelevant; distinct heterogeneous profiles are not. +func checkpointCPUProfiles(cpuinfo string) []string { + profiles := map[string]bool{} + for _, block := range strings.Split(strings.TrimSpace(cpuinfo), "\n\n") { + fields := []string{} + for _, line := range strings.Split(block, "\n") { + key, value, ok := strings.Cut(line, ":") + if !ok { + continue + } + key = strings.TrimSpace(key) + switch key { + case "vendor_id", "cpu family", "model", "stepping", "flags", "Features", "CPU implementer", "CPU architecture", "CPU part", "CPU revision": + value = strings.TrimSpace(value) + if key == "flags" || key == "Features" { + features := strings.Fields(value) + sort.Strings(features) + value = strings.Join(features, " ") + } + fields = append(fields, key+":"+value) + } + } + if len(fields) > 0 { + sort.Strings(fields) + profiles[strings.Join(fields, "\n")] = true + } + } + result := make([]string, 0, len(profiles)) + for profile := range profiles { + result = append(result, profile) + } + sort.Strings(result) + return result +} diff --git a/services/core/internal/sandbox/microsandbox/checkpoint_linux_test.go b/services/core/internal/sandbox/microsandbox/checkpoint_linux_test.go new file mode 100644 index 000000000..0d296776a --- /dev/null +++ b/services/core/internal/sandbox/microsandbox/checkpoint_linux_test.go @@ -0,0 +1,88 @@ +//go:build linux + +package microsandbox + +import ( + "errors" + "os" + "path/filepath" + "reflect" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +func privateCheckpointRoot(t *testing.T, marker string) string { + t.Helper() + root := t.TempDir() + if err := os.Chmod(root, 0700); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(root, ".oac-checkpoint-store"), []byte(marker+"\n"), 0600); err != nil { + t.Fatal(err) + } + return root +} +func TestCheckpointCompatibilityUsesStoreIdentityNotMountOrCachePath(t *testing.T) { + c := testConfig() + c.CheckpointRoot = privateCheckpointRoot(t, "11111111-1111-4111-8111-111111111111") + a, err := CheckpointClass(c) + if err != nil { + t.Fatal(err) + } + c.CheckpointRoot = privateCheckpointRoot(t, "11111111-1111-4111-8111-111111111111") + c.RuntimeHome = "/different/cache" + c.RuntimePath = "/different/msb" + c.FirmwarePath = "/different/firmware" + b, err := CheckpointClass(c) + if err != nil || a != b { + t.Fatal("host paths changed compatibility", a, b, err) + } + c.MemoryMiB++ + b, err = CheckpointClass(c) + if err != nil || a.ExecutionClass == b.ExecutionClass || a.ArtifactDomain != b.ArtifactDomain { + t.Fatal("geometry not fenced", a, b, err) + } + c.CheckpointRoot = privateCheckpointRoot(t, "22222222-2222-4222-8222-222222222222") + b, err = CheckpointClass(c) + if err != nil || a.ArtifactDomain == b.ArtifactDomain { + t.Fatal("foreign store matched", a, b, err) + } +} +func TestCheckpointStoreRequiresPrivateCanonicalIdentity(t *testing.T) { + c := testConfig() + c.CheckpointRoot = privateCheckpointRoot(t, "11111111-1111-4111-8111-111111111111") + if err := os.Chmod(c.CheckpointRoot, 0755); err != nil { + t.Fatal(err) + } + if _, err := CheckpointStore(c); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal(err) + } + if err := os.Chmod(c.CheckpointRoot, 0700); err != nil { + t.Fatal(err) + } + link := filepath.Join(t.TempDir(), "link") + if err := os.Symlink(c.CheckpointRoot, link); err != nil { + t.Fatal(err) + } + c.CheckpointRoot = link + if _, err := CheckpointStore(c); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal(err) + } + c.CheckpointRoot = privateCheckpointRoot(t, "00000000-0000-0000-0000-000000000000") + if _, err := CheckpointStore(c); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal(err) + } +} + +func TestCPUCompatibilityPreservesHeterogeneousProcessorProfiles(t *testing.T) { + first := "processor:0\nmodel:1\nflags:a b\n\nprocessor:1\nmodel:2\nflags:c d\n" + reordered := "processor:9\nflags:d c\nmodel:2\n\nprocessor:4\nflags:b a\nmodel:1\n\nprocessor:5\nflags:a b\nmodel:1\n" + crossed := "processor:0\nmodel:1\nflags:c d\n\nprocessor:1\nmodel:2\nflags:a b\n" + if !reflect.DeepEqual(checkpointCPUProfiles(first), checkpointCPUProfiles(reordered)) { + t.Fatal("processor order/count affected class") + } + if reflect.DeepEqual(checkpointCPUProfiles(first), checkpointCPUProfiles(crossed)) { + t.Fatal("heterogeneous CPU associations collapsed") + } +} diff --git a/services/core/internal/sandbox/microsandbox/checkpoint_other.go b/services/core/internal/sandbox/microsandbox/checkpoint_other.go new file mode 100644 index 000000000..dc92ab8d7 --- /dev/null +++ b/services/core/internal/sandbox/microsandbox/checkpoint_other.go @@ -0,0 +1,10 @@ +//go:build !linux + +package microsandbox + +import "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + +func CheckpointStore(Config) (string, error) { return "", sandbox.ErrHostUnsupported } +func CheckpointClass(Config) (sandbox.CheckpointCompatibility, error) { + return sandbox.CheckpointCompatibility{}, sandbox.ErrHostUnsupported +} diff --git a/services/core/internal/sandbox/microsandbox/identity.go b/services/core/internal/sandbox/microsandbox/identity.go index b831c016d..80ddc2840 100644 --- a/services/core/internal/sandbox/microsandbox/identity.go +++ b/services/core/internal/sandbox/microsandbox/identity.go @@ -24,6 +24,9 @@ func validHash(s string) bool { return e == nil && len(b) == 32 && strings.ToLower(s) == s } func (c Config) Validate() error { + if !filepath.IsAbs(c.CheckpointRoot) || filepath.Clean(c.CheckpointRoot) != c.CheckpointRoot { + return sandbox.ErrInvalid + } if !validID(c.InstallationID) || !filepath.IsAbs(c.HelperPath) || !filepath.IsAbs(c.RuntimeHome) || !filepath.IsAbs(c.RuntimePath) || !filepath.IsAbs(c.FirmwarePath) || !validHash(c.RuntimeSHA256) || !validHash(c.FirmwareSHA256) || c.MemoryMiB == 0 || c.CPUs == 0 || c.RootDiskMiB == 0 || (!c.ExternalWorkspace && c.EnvironmentDiskMiB == 0) { return sandbox.ErrInvalid } @@ -70,7 +73,7 @@ func ValidateCompute(c Config, r sandbox.Reference, v Compute) error { return nil } func ValidateSnapshot(c Config, r sandbox.Reference, s SnapshotIdentity) error { - if !ValidReference(r) || !validID(s.OperationID) || s.Reference != SnapshotReference(c, r, s.OperationID) || s.SourceName != Name(c, r, s.SourceGeneration) || s.SourceID == "" || s.ID == "" || s.Digest == "" || s.CheckpointID == "" || s.CheckpointRoot == "" { + if s.Compatibility.Validate() != nil || !ValidReference(r) || !validID(s.OperationID) || s.Reference != SnapshotReference(c, r, s.OperationID) || s.SourceName != Name(c, r, s.SourceGeneration) || s.SourceID == "" || s.ID == "" || s.Digest == "" || s.CheckpointID == "" || s.CheckpointRoot == "" { return sandbox.ErrInvalid } return nil @@ -99,6 +102,12 @@ func ValidateRequest(q Request) error { if q.Workspace != nil && ((q.Operation != "create" && q.Operation != "resume") || !validID(q.Workspace.ObjectID) || !filepath.IsAbs(q.Workspace.Path) || filepath.Clean(q.Workspace.Path) != q.Workspace.Path || strings.ContainsRune(q.Workspace.Path, 0)) { return sandbox.ErrInvalid } + if q.Workspace != nil { + relative, err := filepath.Rel(q.Workspace.Path, q.Config.CheckpointRoot) + if err != nil || relative == "." || (relative != ".." && !strings.HasPrefix(relative, ".."+string(filepath.Separator))) { + return sandbox.ErrInvalid + } + } switch q.Operation { case "create": if q.Config.ExternalWorkspace != (q.Workspace != nil) { diff --git a/services/core/internal/sandbox/microsandbox/node.go b/services/core/internal/sandbox/microsandbox/node.go index 8f2789053..2b68552dc 100644 --- a/services/core/internal/sandbox/microsandbox/node.go +++ b/services/core/internal/sandbox/microsandbox/node.go @@ -1,6 +1,7 @@ package microsandbox import ( + "context" "errors" "os" "path/filepath" @@ -23,11 +24,12 @@ var NodeArtifacts = []providerassets.Artifact{ // The helper owns local paths; no ambient backend is selected. Resources, the // image and the artifact hashes come from the deployment specification. type Native struct { - HelperPath string `json:"helper_path"` - RuntimeHome string `json:"runtime_home"` - RuntimePath string `json:"runtime_path"` - FirmwarePath string `json:"firmware_path"` - Network Network `json:"network"` + HelperPath string `json:"helper_path"` + RuntimeHome string `json:"runtime_home"` + CheckpointRoot string `json:"checkpoint_root"` + RuntimePath string `json:"runtime_path"` + FirmwarePath string `json:"firmware_path"` + Network Network `json:"network"` } type Network struct { @@ -55,7 +57,7 @@ func configureMicrosandbox(entry Native, spec sandbox.DeploymentSpec, caller *Pr release, resources := spec.Runtime, spec.Resources config := Config{ ExternalWorkspace: spec.Workspace != nil, - InstallationID: result.InstallationID, HelperPath: entry.HelperPath, RuntimeHome: entry.RuntimeHome, RuntimePath: entry.RuntimePath, FirmwarePath: entry.FirmwarePath, + InstallationID: result.InstallationID, HelperPath: entry.HelperPath, RuntimeHome: entry.RuntimeHome, CheckpointRoot: entry.CheckpointRoot, RuntimePath: entry.RuntimePath, FirmwarePath: entry.FirmwarePath, RuntimeSHA256: release.RuntimeSHA256, FirmwareSHA256: release.FirmwareSHA256, Image: release.MicrosandboxRef, MemoryMiB: resources.MemoryMiB, CPUs: uint8(resources.CPUs), RootDiskMiB: resources.RootDiskMiB, EnvironmentDiskMiB: resources.EnvironmentDiskMiB, Network: network, } @@ -65,7 +67,17 @@ func configureMicrosandbox(entry Native, spec sandbox.DeploymentSpec, caller *Pr } provider.workspace = options.Workspace result.Provider = provider - result.Probe = microsandboxProbe(config, resources) + baseProbe := microsandboxProbe(config, resources) + result.Probe = func(ctx context.Context) (*sandbox.CheckpointCompatibility, error) { + if err := baseProbe(ctx); err != nil { + return nil, err + } + c, err := CheckpointClass(config) + if err != nil { + return nil, err + } + return &c, nil + } result.Quiescent = caller.Quiescent result.BackendFingerprint = sandbox.BackendFingerprint("microsandbox", entry.RuntimeHome) return config, nil @@ -75,7 +87,7 @@ func configureMicrosandbox(entry Native, spec sandbox.DeploymentSpec, caller *Pr func BuildNode(c sandbox.NodeConfig, options sandbox.LocalOptions, result *sandbox.Built) (func(), error) { closeProvider := func() {} var entry Native - if sandbox.DecodeConfigurationObject(c.Native, &entry, "helper_path", "runtime_home", "runtime_path", "firmware_path", "network") != nil { + if sandbox.DecodeConfigurationObject(c.Native, &entry, "helper_path", "runtime_home", "runtime_path", "firmware_path", "network", "checkpoint_root") != nil { return closeProvider, errors.New("invalid managed microsandbox node configuration") } caller := &ProcessCaller{} @@ -96,7 +108,15 @@ func BuildNode(c sandbox.NodeConfig, options sandbox.LocalOptions, result *sandb return closeProvider, err } if options.GenerationStateDirectory != "" { - result.Probe = microsandboxGenerationProbe(config, result.Probe) + baseProbe := result.Probe + result.Probe = func(ctx context.Context) (*sandbox.CheckpointCompatibility, error) { + var compatibility *sandbox.CheckpointCompatibility + imageProbe := microsandboxGenerationProbe(config, func(ctx context.Context) error { var err error; compatibility, err = baseProbe(ctx); return err }) + if err := imageProbe(ctx); err != nil { + return nil, err + } + return compatibility, nil + } } return closeProvider, nil } diff --git a/services/core/internal/sandbox/microsandbox/node_test.go b/services/core/internal/sandbox/microsandbox/node_test.go index 082bfd5a7..8e542fda8 100644 --- a/services/core/internal/sandbox/microsandbox/node_test.go +++ b/services/core/internal/sandbox/microsandbox/node_test.go @@ -21,7 +21,13 @@ func TestMicrosandboxConstructionSelectsGenerationReadiness(t *testing.T) { } useKVM(t, true) dir := t.TempDir() - native := Native{HelperPath: filepath.Join(dir, "helper"), RuntimeHome: dir, RuntimePath: filepath.Join(dir, "msb"), FirmwarePath: filepath.Join(dir, "firmware"), Network: Network{DefaultEgress: "allow", DefaultIngress: "deny"}} + native := Native{HelperPath: filepath.Join(dir, "helper"), RuntimeHome: dir, CheckpointRoot: dir, RuntimePath: filepath.Join(dir, "msb"), FirmwarePath: filepath.Join(dir, "firmware"), Network: Network{DefaultEgress: "allow", DefaultIngress: "deny"}} + if err := os.Chmod(dir, 0700); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(dir, ".oac-checkpoint-store"), []byte(uuid.NewString()+"\n"), 0600); err != nil { + t.Fatal(err) + } raw, err := json.Marshal(native) if err != nil { t.Fatal(err) @@ -44,7 +50,7 @@ func TestMicrosandboxConstructionSelectsGenerationReadiness(t *testing.T) { if err != nil { t.Fatal(err) } - err = built.Probe(t.Context()) + _, err = built.Probe(t.Context()) closeProvider() if options.Standalone && err != nil || !options.Standalone && !errors.Is(err, sandbox.ErrRuntimeImageUnavailable) { t.Fatalf("readiness for %+v: %v", options, err) diff --git a/services/core/internal/sandbox/microsandbox/provider.go b/services/core/internal/sandbox/microsandbox/provider.go index 88ce87292..20b093e15 100644 --- a/services/core/internal/sandbox/microsandbox/provider.go +++ b/services/core/internal/sandbox/microsandbox/provider.go @@ -88,6 +88,13 @@ func (p *Provider) state(ctx context.Context, q Request) (State, error) { func (p *Provider) responseState(ctx context.Context, q Request, out Response) (State, error) { var e error + if out.State != nil && out.State.RestoreAttemptClosed != "" { + if q.Operation != "resume" || q.Resume == nil || !out.State.ClosesRestoreAttempt(*q.Resume) { + return State{}, ErrUnconfirmed + } + return *out.State, nil + } + if out.State == nil || ValidateCompute(p.config, q.Reference, out.State.Compute) != nil || out.State.Compute.ID == "" { return State{}, ErrUnconfirmed } diff --git a/services/core/internal/sandbox/microsandbox/provider_test.go b/services/core/internal/sandbox/microsandbox/provider_test.go index 42f955dc1..8b5f1153e 100644 --- a/services/core/internal/sandbox/microsandbox/provider_test.go +++ b/services/core/internal/sandbox/microsandbox/provider_test.go @@ -14,7 +14,7 @@ type callerFunc func(context.Context, Request) (Response, error) func (f callerFunc) Call(c context.Context, q Request) (Response, error) { return f(c, q) } func testConfig() Config { - return Config{ + return Config{CheckpointRoot: "/private-checkpoints", InstallationID: "11111111-1111-4111-8111-111111111111", HelperPath: "/helper", RuntimeHome: "/private/msb", RuntimePath: "/private/bin/msb", FirmwarePath: "/private/lib/libkrunfw.so", RuntimeSHA256: strings.Repeat("a", 64), FirmwareSHA256: strings.Repeat("b", 64), Image: "registry/runtime@sha256:" + strings.Repeat("c", 64), MemoryMiB: 2048, CPUs: 2, RootDiskMiB: 4096, EnvironmentDiskMiB: 2048, Network: NetworkPolicy{DefaultEgress: "deny", DefaultIngress: "deny", Rules: []NetworkRule{{Action: "allow", Direction: "egress", Destination: "host"}}}, @@ -26,7 +26,7 @@ func testRef() sandbox.Reference { func testSnapshot() SnapshotIdentity { c, r := testConfig(), testRef() op := "55555555-5555-4555-8555-555555555555" - return SnapshotIdentity{Reference: SnapshotReference(c, r, op), ID: "snap_exact", Digest: "digest", CheckpointID: "checkpoint", CheckpointRoot: "root", OperationID: op, SourceName: Name(c, r, 0), SourceID: "local:4"} + return SnapshotIdentity{Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "domain", ExecutionClass: "class"}, Reference: SnapshotReference(c, r, op), ID: "snap_exact", Digest: "digest", CheckpointID: "checkpoint", CheckpointRoot: "root", OperationID: op, SourceName: Name(c, r, 0), SourceID: "local:4"} } func deadline(t *testing.T) context.Context { t.Helper() diff --git a/services/core/internal/sandbox/microsandbox/restore_settlement_test.go b/services/core/internal/sandbox/microsandbox/restore_settlement_test.go new file mode 100644 index 000000000..bbf00779b --- /dev/null +++ b/services/core/internal/sandbox/microsandbox/restore_settlement_test.go @@ -0,0 +1,41 @@ +package microsandbox + +import ( + "testing" +) + +func TestRestoreClosedReceiptRequiresExactObserveOnlyOperation(t *testing.T) { + p := &Provider{config: testConfig()} + snapshot := testSnapshot() + target := Compute{Name: Name(testConfig(), testRef(), 1), Generation: 1, RestoredFrom: &snapshot} + q := Request{Operation: "resume", Reference: testRef(), Resume: &ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: target, Snapshot: snapshot, ObserveOnly: true}} + valid := State{Compute: target, Status: "absent", RestoreAttemptClosed: q.Resume.OperationID} + for _, kind := range []string{"exact", "wrong operation", "native ID", "foreign target", "bootstrap", "snapshot", "source stopped", "not observation"} { + t.Run(kind, func(t *testing.T) { + request := q + resume := *q.Resume + request.Resume = &resume + state := valid + switch kind { + case "wrong operation": + state.RestoreAttemptClosed = "wrong" + case "native ID": + state.Compute.ID = "local:1" + case "foreign target": + state.Compute.Name = "foreign" + case "bootstrap": + state.BootstrapComplete = true + case "snapshot": + state.Snapshot = &snapshot + case "source stopped": + state.SourceStopped = true + case "not observation": + request.Resume.ObserveOnly = false + } + _, err := p.responseState(t.Context(), request, Response{State: &state}) + if (kind == "exact") != (err == nil) { + t.Fatal(kind, err) + } + }) + } +} diff --git a/services/core/internal/sandbox/microsandbox/suspension.go b/services/core/internal/sandbox/microsandbox/suspension.go index 46954f196..16dd58719 100644 --- a/services/core/internal/sandbox/microsandbox/suspension.go +++ b/services/core/internal/sandbox/microsandbox/suspension.go @@ -9,7 +9,7 @@ import ( // retained translates a verified native snapshot without exposing its schema to Core. func retained(s SnapshotIdentity) sandbox.RetainedState { raw, _ := json.Marshal(s) - return sandbox.RetainedState{Reference: s.Reference, ID: s.ID, OperationID: s.OperationID, SourceGeneration: s.SourceGeneration, SourceName: s.SourceName, SourceID: s.SourceID, Data: string(raw)} + return sandbox.RetainedState{Compatibility: s.Compatibility, Reference: s.Reference, ID: s.ID, OperationID: s.OperationID, SourceGeneration: s.SourceGeneration, SourceName: s.SourceName, SourceID: s.SourceID, Data: string(raw)} } func (p *Provider) snapshot(r sandbox.Reference, s sandbox.RetainedState) (SnapshotIdentity, error) { var native SnapshotIdentity @@ -41,7 +41,7 @@ func (p *Provider) nativeCompute(r sandbox.Reference, c sandbox.Compute) (Comput return out, nil } func state(s State) sandbox.ComputeState { - out := sandbox.ComputeState{Compute: compute(s.Compute), Status: s.Status, BootstrapComplete: s.BootstrapComplete, ResourcesReleased: s.SourceStopped} + out := sandbox.ComputeState{Compute: compute(s.Compute), Status: s.Status, BootstrapComplete: s.BootstrapComplete, ResourcesReleased: s.SourceStopped, RestoreAttemptClosed: s.RestoreAttemptClosed} if s.Snapshot != nil { v := retained(*s.Snapshot) out.Retained = &v @@ -89,15 +89,6 @@ func (p *Provider) Suspend(ctx context.Context, q sandbox.SuspendRequest) (sandb n.Snapshot = &s } s, e := p.nativeSuspend(ctx, n) - if e == nil && s.Snapshot != nil { - // Complete the same exact-source cleanup previously performed by Core. - // Native KillCompute still verifies ownership and removes its disks. - e = p.nativeKillCompute(ctx, q.Reference, c) - if e == nil { - s.SourceStopped = true - s.Status = "suspended" - } - } out := state(s) // The helper holds the allocation lock until the original native work settles. out.SuspendSettled = e == nil diff --git a/services/core/internal/sandbox/microsandbox/suspension_contract_test.go b/services/core/internal/sandbox/microsandbox/suspension_contract_test.go index 284bc2159..ce0bb688a 100644 --- a/services/core/internal/sandbox/microsandbox/suspension_contract_test.go +++ b/services/core/internal/sandbox/microsandbox/suspension_contract_test.go @@ -2,7 +2,7 @@ package microsandbox import ( "context" - "errors" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "testing" ) @@ -19,26 +19,21 @@ func TestSharedSuspendSettlesExactSourceCleanupInsideAdapter(t *testing.T) { if q.Suspend.ObserveOnly != reconcile { t.Fatal("reconcile intent changed") } - return Response{Version: ProtocolVersion, State: &State{Compute: source, Status: "suspended", BootstrapComplete: true, Snapshot: &snap, SourceStopped: !reconcile}}, nil - } - if q.Operation != "kill" || q.Compute.ID != source.ID { - t.Fatal("wrong source cleanup", q.Operation) - } - if cleanupFails { - return Response{}, errors.New("lost cleanup") + return Response{Version: ProtocolVersion, State: &State{Compute: source, Status: "suspended", BootstrapComplete: true, Snapshot: &snap, SourceStopped: !cleanupFails}}, nil } - return Response{Version: ProtocolVersion}, nil + t.Fatal("unexpected extra native operation", q.Operation) + return Response{}, nil })) if e != nil { t.Fatal(e) } q := sandbox.SuspendRequest{Reference: r, OperationID: snap.OperationID, Source: compute(source), ReconcileOnly: reconcile} got, e := p.Suspend(deadline(t), q) - if len(calls) != 2 || calls[0] != "suspend" || calls[1] != "kill" { + if len(calls) != 1 || calls[0] != "suspend" { t.Fatal(calls) } if cleanupFails { - if e == nil || got.SuspendSettled { + if got.ResourcesReleased || (e == nil && sandbox.ValidateSuspendResult(q, got) == nil) { t.Fatal("failed cleanup released capacity") } continue diff --git a/services/core/internal/sandbox/microsandbox/types.go b/services/core/internal/sandbox/microsandbox/types.go index a947b3b47..012a9f56a 100644 --- a/services/core/internal/sandbox/microsandbox/types.go +++ b/services/core/internal/sandbox/microsandbox/types.go @@ -11,7 +11,7 @@ import ( ) const ProtocolVersion = 5 -const SDKVersion = "v0.7.2" +const SDKVersion = "v0.7.8" const MaxOutputBytes = 1024 * 1024 const MaxRequestBytes = 72 * 1024 * 1024 const MaxResponseBytes = 16 * 1024 * 1024 @@ -24,6 +24,7 @@ type Config struct { InstallationID string HelperPath string RuntimeHome string + CheckpointRoot string RuntimePath string FirmwarePath string RuntimeSHA256 string @@ -55,6 +56,7 @@ type Compute struct { // persists it unchanged and records consumption separately; it never invents // paths, checksums, native checkpoint fields, or source identity. type SnapshotIdentity struct { + Compatibility sandbox.CheckpointCompatibility Reference string ID string Digest string @@ -67,11 +69,12 @@ type SnapshotIdentity struct { } type State struct { - Compute Compute - Status string - BootstrapComplete bool - Snapshot *SnapshotIdentity - SourceStopped bool + RestoreAttemptClosed string + Compute Compute + Status string + BootstrapComplete bool + Snapshot *SnapshotIdentity + SourceStopped bool } type SuspendRequest struct { Reference sandbox.Reference @@ -138,3 +141,11 @@ type Metrics struct { type Caller interface { Call(context.Context, Request) (Response, error) } + +// ClosesRestoreAttempt verifies the exact never-dispatched private helper attempt. +func (s State) ClosesRestoreAttempt(q ResumeRequest) bool { + return q.ObserveOnly && q.OperationID != "" && s.RestoreAttemptClosed == q.OperationID && s.Status == "absent" && + s.Compute.ID == "" && q.Target.ID == "" && s.Compute.Name == q.Target.Name && s.Compute.Generation == q.Target.Generation && + s.Compute.RestoredFrom != nil && q.Target.RestoredFrom != nil && *s.Compute.RestoredFrom == q.Snapshot && *q.Target.RestoredFrom == q.Snapshot && + !s.BootstrapComplete && !s.SourceStopped && s.Snapshot == nil +} diff --git a/services/core/internal/sandbox/node/checkpoint_generation_test.go b/services/core/internal/sandbox/node/checkpoint_generation_test.go new file mode 100644 index 000000000..b54341383 --- /dev/null +++ b/services/core/internal/sandbox/node/checkpoint_generation_test.go @@ -0,0 +1,98 @@ +package node + +import ( + "context" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/providercontract" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" + "github.com/google/uuid" + "net/http/httptest" + "strings" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +func TestStaticNodeReportsExactCheckpointGeneration(t *testing.T) { + id := identity() + compatibility := &sandbox.CheckpointCompatibility{ArtifactDomain: "private-store", ExecutionClass: "native-class"} + status := sandbox.GenerationStatus{Generation: id.DeploymentGeneration, SpecificationDigest: id.SpecificationDigest, State: "ready", Checkpoint: compatibility} + seen := false + hub := NewHub(HubOptions{Heartbeat: func(_ context.Context, _ Identity, _ string, _ uint64, health Health) error { + seen = true + if len(health.Generations) != 1 || *health.Generations[0].Checkpoint != *compatibility { + t.Fatal("qualification lost") + } + return nil + }}) + p := &peer{identity: id} + health := Health{ProviderReady: true, Generations: []sandbox.GenerationStatus{status}} + if err := hub.recordHealth(t.Context(), p, health); err != nil || !seen { + t.Fatal(err, seen) + } + for _, mutation := range []string{"generation", "digest", "readiness", "extra"} { + t.Run(mutation, func(t *testing.T) { + changed := health + changed.Generations = append([]sandbox.GenerationStatus(nil), health.Generations...) + switch mutation { + case "generation": + changed.Generations[0].Generation++ + case "digest": + changed.Generations[0].SpecificationDigest = strings.Repeat("f", 64) + case "readiness": + changed.ProviderReady = false + case "extra": + changed.Generations = append(changed.Generations, status) + } + seen = false + if err := hub.recordHealth(t.Context(), p, changed); err == nil || seen { + t.Fatal("invalid static qualification reached persistence") + } + }) + } +} + +type closedRestoreProvider struct{ fakeProvider } + +func (*closedRestoreProvider) ProviderOperations() providercontract.Operations { + return microsandbox.Operations() +} +func (*closedRestoreProvider) Resume(_ context.Context, q sandbox.ResumeRequest) (sandbox.ComputeState, error) { + return sandbox.ComputeState{Compute: q.Target, Status: "absent", RestoreAttemptClosed: q.OperationID}, nil +} +func TestClosedRestoreEvidenceSurvivesNodeTransport(t *testing.T) { + id := identity() + var credential string + hub := NewHub(HubOptions{Authenticate: func(_ context.Context, node, token string) (Identity, error) { + if node != id.NodeID || token != credential { + return Identity{}, ErrAuthentication + } + return id, nil + }, OwnerEpoch: func(context.Context) (uint64, error) { return 7, nil }}) + defer hub.Close() + server := httptest.NewServer(hub) + defer server.Close() + directory := stateDir(t) + stored, err := InitIdentity(directory, server.URL, id) + if err != nil { + t.Fatal(err) + } + credential = stored.Credential + ctx, cancel := context.WithCancel(t.Context()) + done := make(chan error, 1) + go func() { + done <- Run(ctx, AgentConfig{CoreURL: server.URL, StateDirectory: directory, Identity: id, Credential: credential, Provider: &closedRestoreProvider{}, Probe: probe}) + }() + defer func() { + cancel() + if err := <-done; err != nil { + t.Error(err) + } + }() + wait(t, func() bool { return hub.Online(id.NodeID) }) + snapshot := sandbox.RetainedState{ID: "snapshot", Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "store", ExecutionClass: "class"}} + q := sandbox.ResumeRequest{Reference: reference(), OperationID: uuid.NewString(), Retained: snapshot, Target: sandbox.Compute{Generation: 1, Name: "exact-target", RestoredFrom: &snapshot}, ReconcileOnly: true} + result, err := hub.Proxy(id.NodeID, microsandbox.Operations(), id.DeploymentGeneration).Resume(t.Context(), q) + if err != nil || !result.ClosesRestoreAttempt(q) { + t.Fatalf("exact closed proof lost in proxy: %#v %v", result, err) + } +} diff --git a/services/core/internal/sandbox/node/generation_json.go b/services/core/internal/sandbox/node/generation_json.go index 75e85ea36..d3a68ef7b 100644 --- a/services/core/internal/sandbox/node/generation_json.go +++ b/services/core/internal/sandbox/node/generation_json.go @@ -62,9 +62,15 @@ func generationRecords(raw []byte, required, optional string) error { return sandbox.ErrInvalid } for _, entry := range entries { - if _, err := generationObject(entry, required, optional, ""); err != nil { + fields, err := generationObject(entry, required, optional, "") + if err != nil { return err } + if checkpoint := fields["checkpoint"]; checkpoint != nil { + if _, err := generationObject(checkpoint, "artifact_domain execution_class", "", ""); err != nil { + return err + } + } } return nil } @@ -110,7 +116,7 @@ func validateGenerationJSON(raw []byte, kind string) error { if err != nil { return err } - if err := generationRecords(health["generations"], "generation specification_digest state", "diagnostic"); err != nil { + if err := generationRecords(health["generations"], "generation specification_digest state", "diagnostic checkpoint"); err != nil { return err } } diff --git a/services/core/internal/sandbox/node/generation_json_test.go b/services/core/internal/sandbox/node/generation_json_test.go index 8fa769f8e..37fbd8a3f 100644 --- a/services/core/internal/sandbox/node/generation_json_test.go +++ b/services/core/internal/sandbox/node/generation_json_test.go @@ -2,6 +2,7 @@ package node import ( "encoding/json" + "fmt" "strings" "testing" "time" @@ -63,7 +64,7 @@ func TestGenerationHealthRejectsUnboundedOrAmbiguousNumbers(t *testing.T) { } func TestNodeProtocolRejectsHistoricalVersions(t *testing.T) { - for _, version := range []int{1, 2, 3, 4, 5, 6} { + for _, version := range []int{1, 2, 3, 4, 5, 6, 7} { raw, _ := json.Marshal(frame{Version: version, Type: "hello", Identity: new(Identity), Health: &Health{ObservedAt: time.Now().UTC()}}) if _, err := decodeFrame(raw); err == nil { t.Fatalf("accepted historical protocol %d", version) @@ -84,8 +85,9 @@ func TestNodeGenerationManagementIsAnExplicitCurrentCapability(t *testing.T) { if managed { f.GenerationManagement = false raw, _ = json.Marshal(f) - if _, err := decodeFrame(raw); err == nil { - t.Fatal("generation management accepted without declaration") + decoded, err := decodeFrame(raw) + if err != nil || decoded.GenerationManagement || len(decoded.Health.Generations) != 1 { + t.Fatal("static qualification implicitly enabled generation management", err) } } } @@ -110,3 +112,59 @@ func TestNodeBootstrapRejectsAmbiguousHarness(t *testing.T) { t.Fatal("missing Harness accepted") } } + +func TestCheckpointGenerationHealthJSONRoundTrip(t *testing.T) { + for _, managed := range []bool{false, true} { + for _, kind := range []string{"hello", "heartbeat"} { + t.Run(fmt.Sprintf("managed=%t/%s", managed, kind), func(t *testing.T) { + compatibility := sandbox.CheckpointCompatibility{ArtifactDomain: "private-store", ExecutionClass: "native-class"} + f := frame{Version: ProtocolVersion, Type: kind, Health: &Health{ProviderReady: true, ObservedAt: time.Now().UTC(), Generations: []sandbox.GenerationStatus{{Generation: 5, SpecificationDigest: strings.Repeat("a", 64), State: "ready", Checkpoint: &compatibility}}}} + if kind == "hello" { + f.Identity = new(Identity) + f.GenerationManagement = managed + } else { + f.ConnectionID = uuid.NewString() + f.OwnerEpoch = 1 + } + raw, err := json.Marshal(f) + if err != nil { + t.Fatal(err) + } + if err := validateVersionFrame(f, len(raw)); err != nil { + t.Fatal("writer rejected valid status", err) + } + decoded, err := decodeFrame(raw) + if err != nil { + t.Fatal("decoder rejected valid status", err) + } + if decoded.Health == nil || len(decoded.Health.Generations) != 1 || decoded.Health.Generations[0].Checkpoint == nil || *decoded.Health.Generations[0].Checkpoint != compatibility { + t.Fatal("checkpoint qualification lost", decoded) + } + original := string(raw) + for name, replacement := range map[string]string{ + "unknown nested field": strings.Replace(original, `"artifact_domain":"private-store"`, `"artifact_domain":"private-store","extra":"value"`, 1), + "unknown generation field": strings.Replace(original, `"checkpoint":`, `"extra":true,"checkpoint":`, 1), + "duplicate token": strings.Replace(original, `"artifact_domain":"private-store"`, `"artifact_domain":"private-store","artifact_domain":"other"`, 1), + "case alias": strings.Replace(original, `"artifact_domain":`, `"Artifact_Domain":`, 1), + "missing token": strings.Replace(original, `"artifact_domain":"private-store",`, "", 1), + "null token": strings.Replace(original, `"artifact_domain":"private-store"`, `"artifact_domain":null`, 1), + "null checkpoint": strings.Replace(original, `{"artifact_domain":"private-store","execution_class":"native-class"}`, `null`, 1), + "empty domain": strings.Replace(original, `"private-store"`, `""`, 1), + "empty execution": strings.Replace(original, `"native-class"`, `""`, 1), + "oversized domain": strings.Replace(original, `"private-store"`, `"`+strings.Repeat("x", 257)+`"`, 1), + "oversized execution": strings.Replace(original, `"native-class"`, `"`+strings.Repeat("x", 257)+`"`, 1), + "nonready checkpoint": strings.Replace(original, `"state":"ready"`, `"state":"preparing"`, 1), + } { + t.Run(name, func(t *testing.T) { + if replacement == original { + t.Fatal("fixture unchanged") + } + if _, err := decodeFrame([]byte(replacement)); err == nil { + t.Fatal("invalid checkpoint accepted") + } + }) + } + }) + } + } +} diff --git a/services/core/internal/sandbox/node/generation_wire.go b/services/core/internal/sandbox/node/generation_wire.go index 11884877b..d8c83327b 100644 --- a/services/core/internal/sandbox/node/generation_wire.go +++ b/services/core/internal/sandbox/node/generation_wire.go @@ -46,6 +46,9 @@ func validateVersionFrame(f frame, size int) error { if !validGeneration(g.Generation) || !validSpecificationDigest(g.SpecificationDigest) || seen[g.Generation] || (g.State != "ready" && g.State != "preparing" && g.State != "failed") || g.State == "ready" && g.Diagnostic != "" || len(g.Diagnostic) > 64 || sandbox.NormalizeNodeDiagnostic(g.Diagnostic) != g.Diagnostic { return sandbox.ErrInvalid } + if g.Checkpoint != nil && (g.State != "ready" || g.Checkpoint.Validate() != nil) { + return sandbox.ErrInvalid + } seen[g.Generation] = true } } @@ -72,9 +75,6 @@ func validateVersionFrame(f frame, size int) error { } switch f.Type { case "hello": - if !f.GenerationManagement && f.Health != nil && f.Health.Generations != nil { - return sandbox.ErrInvalid - } if f.Identity == nil || f.Health == nil || f.Request != nil || f.Response != nil || f.Control != nil || f.Deployment != nil { return sandbox.ErrInvalid } diff --git a/services/core/internal/sandbox/node/generations.go b/services/core/internal/sandbox/node/generations.go index 83354a61a..c28029a8d 100644 --- a/services/core/internal/sandbox/node/generations.go +++ b/services/core/internal/sandbox/node/generations.go @@ -15,7 +15,7 @@ type GenerationProvider struct { Generation uint64 SpecificationDigest string Provider sandbox.SandboxProvider - Probe func(context.Context) error + Probe func(context.Context) (*sandbox.CheckpointCompatibility, error) // Quiescent reports that no helper outlived its canceled caller; nil // means always quiescent. Quiescent func() bool @@ -31,6 +31,7 @@ type GenerationManagerOptions struct { } type localGeneration struct { + checkpoint *sandbox.CheckpointCompatibility value GenerationProvider state, diagnostic string refs int @@ -164,7 +165,7 @@ func (m *GenerationManager) Statuses() []sandbox.GenerationStatus { continue } seen[key] = true - result = append(result, sandbox.GenerationStatus{Generation: key, SpecificationDigest: g.value.SpecificationDigest, State: g.state, Diagnostic: g.diagnostic}) + result = append(result, sandbox.GenerationStatus{Generation: key, SpecificationDigest: g.value.SpecificationDigest, State: g.state, Diagnostic: g.diagnostic, Checkpoint: g.checkpoint}) if key != m.target.Generation && (m.target.ServingGeneration == nil || key != *m.target.ServingGeneration) { m.statusCursor = key } @@ -360,7 +361,10 @@ func (m *GenerationManager) probeLoop() { continue } ctx, cancel := context.WithTimeout(m.ctx, 5*time.Second) - err := g.value.Probe(ctx) + checkpoint, err := g.value.Probe(ctx) + if err == nil && ((sandbox.SupportsSuspension(g.value.Provider) && (checkpoint == nil || checkpoint.Validate() != nil)) || (!sandbox.SupportsSuspension(g.value.Provider) && checkpoint != nil)) { + err = sandbox.ErrInvalid + } cancel() m.mu.Lock() g.refs-- @@ -368,9 +372,11 @@ func (m *GenerationManager) probeLoop() { if !errors.Is(err, context.Canceled) { state, diagnostic := g.state, g.diagnostic g.state = "ready" + g.checkpoint = checkpoint g.diagnostic = "" if err != nil { g.state = "failed" + g.checkpoint = nil g.diagnostic = sandbox.NodeDiagnostic(err) if errors.Is(err, sandbox.ErrRuntimeImageUnavailable) || errors.Is(err, sandbox.ErrArtifactsUnavailable) { g.repairing = true diff --git a/services/core/internal/sandbox/node/generations_test.go b/services/core/internal/sandbox/node/generations_test.go index bce4d88cd..f81a5e38f 100644 --- a/services/core/internal/sandbox/node/generations_test.go +++ b/services/core/internal/sandbox/node/generations_test.go @@ -207,7 +207,7 @@ func TestRestartRecoversExactOlderGenerationBeforeAdvertisingReadiness(t *testin probe := make(chan struct{}, 1) qualify := make(chan struct{}) m, err := NewGenerationManager(t.Context(), GenerationManagerOptions{ - Initial: []GenerationProvider{{Generation: 2, SpecificationDigest: digest, Provider: &fakeProvider{}, Probe: func(context.Context) error { return nil }}}, + Initial: []GenerationProvider{{Generation: 2, SpecificationDigest: digest, Provider: &fakeProvider{}, Probe: func(context.Context) (*sandbox.CheckpointCompatibility, error) { return nil, nil }}}, Recover: []sandbox.GenerationReference{{Generation: 1, SpecificationDigest: digest}}, Prepare: func(ctx context.Context, generation uint64, got string) (GenerationProvider, error) { if got != digest { @@ -219,16 +219,16 @@ func TestRestartRecoversExactOlderGenerationBeforeAdvertisingReadiness(t *testin case <-ctx.Done(): return GenerationProvider{}, ctx.Err() } - return GenerationProvider{Generation: generation, SpecificationDigest: got, Provider: &fakeProvider{}, Probe: func(ctx context.Context) error { + return GenerationProvider{Generation: generation, SpecificationDigest: got, Provider: &fakeProvider{}, Probe: func(ctx context.Context) (*sandbox.CheckpointCompatibility, error) { select { case probe <- struct{}{}: default: } select { case <-qualify: - return nil + return nil, nil case <-ctx.Done(): - return ctx.Err() + return nil, ctx.Err() } }}, nil }, @@ -296,12 +296,12 @@ func TestInterruptedCollectionNeverPreparesOrServesAfterRestart(t *testing.T) { probed := make(chan struct{}, 1) removes := 0 manager, err := NewGenerationManager(t.Context(), GenerationManagerOptions{ - Initial: []GenerationProvider{{Generation: 2, SpecificationDigest: digest, Provider: &fakeProvider{}, Probe: func(context.Context) error { + Initial: []GenerationProvider{{Generation: 2, SpecificationDigest: digest, Provider: &fakeProvider{}, Probe: func(context.Context) (*sandbox.CheckpointCompatibility, error) { select { case probed <- struct{}{}: default: } - return nil + return nil, nil }}}, Collect: []sandbox.GenerationReference{ref}, Prepare: func(context.Context, uint64, string) (GenerationProvider, error) { diff --git a/services/core/internal/sandbox/node/hub.go b/services/core/internal/sandbox/node/hub.go index bda1da1c4..4feb0e2f6 100644 --- a/services/core/internal/sandbox/node/hub.go +++ b/services/core/internal/sandbox/node/hub.go @@ -391,8 +391,10 @@ func (h *Hub) call(ctx context.Context, id string, q request) (response, error) } func (h *Hub) recordHealth(ctx context.Context, p *peer, health Health) error { - if !p.generationManagement && health.Generations != nil { - return sandbox.ErrInvalid + if !p.generationManagement && len(health.Generations) > 0 { + if len(health.Generations) != 1 || health.Generations[0].Generation != p.identity.DeploymentGeneration || health.Generations[0].SpecificationDigest != p.identity.SpecificationDigest || (health.Generations[0].State == "ready") != health.ProviderReady { + return sandbox.ErrInvalid + } } if p.generationManagement { if h.options.Generations == nil { diff --git a/services/core/internal/sandbox/node/proxy.go b/services/core/internal/sandbox/node/proxy.go index 9271fb418..c345a50c1 100644 --- a/services/core/internal/sandbox/node/proxy.go +++ b/services/core/internal/sandbox/node/proxy.go @@ -123,6 +123,12 @@ func (p *provider) state(ctx context.Context, q request) (sandbox.ComputeState, } return sandbox.ComputeState{}, e } + if r.State != nil && r.State.RestoreAttemptClosed != "" { + if q.Operation == "resume" && q.Resume != nil && r.State.ClosesRestoreAttempt(*q.Resume) { + return *r.State, nil + } + return sandbox.ComputeState{}, sandbox.ErrComputeUnconfirmed + } if r.State == nil || r.State.Compute.ID == "" { return sandbox.ComputeState{}, sandbox.ErrComputeUnconfirmed } diff --git a/services/core/internal/sandbox/node/wire.go b/services/core/internal/sandbox/node/wire.go index 58d51b97d..31bfeefa6 100644 --- a/services/core/internal/sandbox/node/wire.go +++ b/services/core/internal/sandbox/node/wire.go @@ -19,7 +19,7 @@ import ( "github.com/gorilla/websocket" ) -const ProtocolVersion = 7 +const ProtocolVersion = 8 const MaxControlFrameBytes = 32 * 1024 const MaxFrameBytes = 72 * 1024 * 1024 const maxPending = 32 diff --git a/services/core/internal/sandbox/node/workspace_test.go b/services/core/internal/sandbox/node/workspace_test.go index e144224bd..3c4ff1f7a 100644 --- a/services/core/internal/sandbox/node/workspace_test.go +++ b/services/core/internal/sandbox/node/workspace_test.go @@ -30,22 +30,29 @@ func TestWorkspaceWireBindsConfigurationAndEnvironment(t *testing.T) { case "object": b.Attachment.Reference.ObjectID = "invalid" } - q := request{DeploymentGeneration: 1, Sequence: 1, OwnerEpoch: 1, ConnectionID: ref.TenantID, ID: ref.AllocationID, Reference: ref, Operation: "create", TimeoutMillis: 1000, Bootstrap: &sandbox.Bootstrap{Reference: ref, Harness: "codex", Workspace: &b}} + q := request{ID: ref.AllocationID, Reference: ref, Operation: "create", TimeoutMillis: 1000, Bootstrap: &sandbox.Bootstrap{Harness: "codex", Reference: ref, Workspace: &b}} if (q.validate() == nil) != (fault == "match") { t.Fatal("invalid create binding forwarding", fault) } - raw, err := json.Marshal(frame{Version: ProtocolVersion, Type: "request", Request: &q}) - if err != nil { - t.Fatal(err) + for _, external := range []bool{false, true} { + wireRequest := q + bootstrap := *q.Bootstrap + wireRequest.Bootstrap = &bootstrap + if !external { + bootstrap.Workspace = nil + } + wireRequest.DeploymentGeneration, wireRequest.Sequence, wireRequest.OwnerEpoch = 1, 1, 1 + wireRequest.ConnectionID = ref.TenantID + raw, err := json.Marshal(frame{Version: ProtocolVersion, Type: "request", Request: &wireRequest}) + if err != nil { + t.Fatal(err) + } + decoded, err := decodeFrame(raw) + want := !external || fault == "match" + if (err == nil && decoded.Request.validate() == nil) != want { + t.Fatalf("workspace wire external=%v fault=%s: %v", external, fault, err) + } } - decoded, err := decodeFrame(raw) - if err != nil { - t.Fatal("workspace bootstrap rejected by strict wire", err) - } - if (decoded.Request.validate() == nil) != (fault == "match") { - t.Fatal("workspace binding lost on wire", fault) - } - q.Operation = "resume" q.Bootstrap = nil q.Resume = &sandbox.ResumeRequest{Reference: ref, Workspace: &b} @@ -80,14 +87,14 @@ func (c *workspaceRefusalCaller) Call(context.Context, microsandbox.Request) (mi func TestActualWorkspaceRefusalKeepsSettledAbsenceOnWire(t *testing.T) { ref := reference() caller := new(workspaceRefusalCaller) - config := microsandbox.Config{ExternalWorkspace: true, InstallationID: "11111111-1111-4111-8111-111111111111", HelperPath: "/helper", RuntimeHome: "/runtime", RuntimePath: "/runtime/msb", FirmwarePath: "/runtime/firmware", RuntimeSHA256: strings.Repeat("a", 64), FirmwareSHA256: strings.Repeat("b", 64), Image: "image@sha256:" + strings.Repeat("c", 64), CPUs: 2, MemoryMiB: 2048, RootDiskMiB: 4096, Network: microsandbox.NetworkPolicy{DefaultEgress: "deny", DefaultIngress: "deny"}} + config := microsandbox.Config{ExternalWorkspace: true, InstallationID: "11111111-1111-4111-8111-111111111111", HelperPath: "/helper", RuntimeHome: "/runtime", CheckpointRoot: "/checkpoints", RuntimePath: "/runtime/msb", FirmwarePath: "/runtime/firmware", RuntimeSHA256: strings.Repeat("a", 64), FirmwareSHA256: strings.Repeat("b", 64), Image: "image@sha256:" + strings.Repeat("c", 64), CPUs: 2, MemoryMiB: 2048, RootDiskMiB: 4096, Network: microsandbox.NetworkPolicy{DefaultEgress: "deny", DefaultIngress: "deny"}} provider, err := microsandbox.NewWithCaller(config, caller) if err != nil { t.Fatal(err) } fsConfig := workspacefs.Configuration{ID: "44444444-4444-4444-8444-444444444444", Adapter: "fixture", Parameters: json.RawMessage(`{}`)} binding := &workspacefs.Binding{Configuration: fsConfig, Attachment: workspacefs.Attachment{Reference: workspacefs.Reference{TenantID: ref.TenantID, EnvironmentID: ref.EnvironmentID, ObjectID: "55555555-5555-4555-8555-555555555555"}, ConfigurationID: fsConfig.ID, Kind: workspacefs.AttachmentHostDirectory, Native: json.RawMessage(`{}`)}} - bootstrap := sandbox.Bootstrap{Reference: ref, SessionID: ref.TenantID, DeviceID: ref.EnvironmentID, CoreURL: "https://core.example/api/v1", Credential: "fixture", NetworkAccess: "disabled", Harness: "codex", Workspace: binding} + bootstrap := sandbox.Bootstrap{Harness: "codex", Reference: ref, SessionID: ref.TenantID, DeviceID: ref.EnvironmentID, CoreURL: "https://core.example/api/v1", Credential: "fixture", NetworkAccess: "disabled", Workspace: binding} ctx, cancel := context.WithTimeout(t.Context(), time.Second) defer cancel() out := execute(ctx, provider, request{ID: ref.AllocationID, ConnectionID: ref.TenantID, Reference: ref, Operation: "create", TimeoutMillis: 1000, Bootstrap: &bootstrap}) diff --git a/services/core/internal/sandbox/providers/configuration_flow_test.go b/services/core/internal/sandbox/providers/configuration_flow_test.go index abc454120..3d87a8926 100644 --- a/services/core/internal/sandbox/providers/configuration_flow_test.go +++ b/services/core/internal/sandbox/providers/configuration_flow_test.go @@ -4,6 +4,7 @@ import ( "bytes" "context" "encoding/json" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "net/http/httptest" "strings" "testing" @@ -115,7 +116,7 @@ func TestAdditionalConfigurationProviderUsesCommonAPIAndStore(t *testing.T) { t.Fatal(err) } defer lease.Close(context.Background()) - changes, err := deployment.NewExecutionOperations(service, deploymentpg.NewExecution(lease, pgtest.CredentialKey(t))) + changes, err := deployment.NewExecutionOperations(service, deploymentpg.NewExecution(lease, pgtest.CredentialKey(t)), engine.Catalog{}) if err != nil { t.Fatal(err) } diff --git a/services/core/internal/sandbox/providers/generation_test.go b/services/core/internal/sandbox/providers/generation_test.go index 0bacbe40d..60b42d297 100644 --- a/services/core/internal/sandbox/providers/generation_test.go +++ b/services/core/internal/sandbox/providers/generation_test.go @@ -21,7 +21,7 @@ func generationConfig(t *testing.T) (sandbox.NodeConfig, sandboxmicro.Native) { dir := t.TempDir() spec := validRegistrationSpec() spec.Resources.RootDiskMiB, spec.Resources.EnvironmentDiskMiB = 8192, 8192 - native := sandboxmicro.Native{HelperPath: filepath.Join(dir, "helper"), RuntimeHome: dir, RuntimePath: filepath.Join(dir, "msb"), FirmwarePath: filepath.Join(dir, "firmware"), + native := sandboxmicro.Native{HelperPath: filepath.Join(dir, "helper"), RuntimeHome: dir, CheckpointRoot: t.TempDir(), RuntimePath: filepath.Join(dir, "msb"), FirmwarePath: filepath.Join(dir, "firmware"), Network: sandboxmicro.Network{DefaultEgress: "allow", DefaultIngress: "deny"}} raw, err := json.Marshal(native) if err != nil { diff --git a/services/core/internal/sandbox/retained_state_test.go b/services/core/internal/sandbox/retained_state_test.go new file mode 100644 index 000000000..c76f10010 --- /dev/null +++ b/services/core/internal/sandbox/retained_state_test.go @@ -0,0 +1,51 @@ +package sandbox + +import ( + "encoding/json" + "testing" +) + +func TestRetainedCompatibilityIsOptionalButNeverPartial(t *testing.T) { + legacy := `{"Reference":"native","ID":"one","Data":"opaque","OperationID":"capture","SourceGeneration":1,"SourceName":"source","SourceID":"one"}` + var retained RetainedState + if err := json.Unmarshal([]byte(legacy), &retained); err != nil || ValidateRetained(retained) != nil { + t.Fatal("legacy native pause envelope rejected", err) + } + for _, compatibility := range []CheckpointCompatibility{{}, {ArtifactDomain: "store"}, {ExecutionClass: "class"}, {ArtifactDomain: "store", ExecutionClass: "class"}} { + retained.Compatibility = compatibility + wantValid := compatibility == (CheckpointCompatibility{}) || compatibility.Validate() == nil + if (ValidateRetained(retained) == nil) != wantValid { + t.Fatal("partial compatibility accepted", compatibility) + } + } +} + +func TestRetainedClosureCannotAuthorizeAnotherRestore(t *testing.T) { + retained := RetainedState{Reference: "native", ID: "one", Data: "opaque", OperationID: "capture", SourceGeneration: 1, SourceName: "source", SourceID: "one"} + target := Compute{Generation: 2, Name: "target", RestoredFrom: &retained} + q := ResumeRequest{OperationID: "restore", Retained: retained, Target: target, ReconcileOnly: true} + good := ComputeState{Compute: target, Status: "absent", RestoreAttemptClosed: "restore"} + if !good.ClosesRestoreAttempt(q) { + t.Fatal("exact settled closure rejected") + } + for _, kind := range []string{"operation", "generation", "provenance", "dispatched", "fresh"} { + s, request := good, q + switch kind { + case "operation": + s.RestoreAttemptClosed = "other" + case "generation": + s.Compute.Generation++ + case "provenance": + other := retained + other.Data = "foreign" + s.Compute.RestoredFrom = &other + case "dispatched": + s.Compute.ID = "native-target" + case "fresh": + request.ReconcileOnly = false + } + if s.ClosesRestoreAttempt(request) { + t.Fatal("unbound closure accepted", kind) + } + } +} diff --git a/services/core/internal/sandbox/sandbox_provider.go b/services/core/internal/sandbox/sandbox_provider.go index b95182b6e..211a6053d 100644 --- a/services/core/internal/sandbox/sandbox_provider.go +++ b/services/core/internal/sandbox/sandbox_provider.go @@ -134,6 +134,9 @@ var suspensionOperations = []string{"Initial", "NewCompute", "GetCompute", "Rene // ValidateRetained checks the shared envelope; only its adapter interprets Data. func ValidateRetained(s RetainedState) error { + if s.Compatibility != (CheckpointCompatibility{}) && s.Compatibility.Validate() != nil { + return ErrInvalid + } if s.Reference == "" || s.ID == "" || s.OperationID == "" || s.SourceID == "" || s.SourceName == "" || len(s.Data) == 0 || len(s.Data) > 64*1024 { return ErrInvalid } @@ -149,9 +152,24 @@ type Compute struct { RestoredFrom *RetainedState } -// RetainedState is adapter-owned recoverable state. Data is opaque to Core. -// A retained state does not imply an independent snapshot. +// CheckpointCompatibility identifies an adapter-verified private archive domain +// and execution compatibility class. Core compares these opaque tokens only. +type CheckpointCompatibility struct { + ArtifactDomain string `json:"artifact_domain"` + ExecutionClass string `json:"execution_class"` +} + +func (c CheckpointCompatibility) Validate() error { + if c.ArtifactDomain == "" || c.ExecutionClass == "" || len(c.ArtifactDomain) > 256 || len(c.ExecutionClass) > 256 { + return ErrInvalid + } + return nil +} + +// RetainedState is adapter-owned recoverable state; Data stays opaque to Core. +// Optional Compatibility declares portable checkpoint qualification. type RetainedState struct { + Compatibility CheckpointCompatibility `json:",omitzero"` Reference string ID string Data string @@ -162,13 +180,26 @@ type RetainedState struct { } type ComputeState struct { - Compute Compute - Status string - BootstrapComplete bool - Retained *RetainedState - ResourcesReleased bool - SuspendSettled bool + RestoreAttemptClosed string + Compute Compute + Status string + BootstrapComplete bool + Retained *RetainedState + ResourcesReleased bool + // SuspendSettled without Retained also durably fences every late dispatch + // of this exact source and operation; native absence alone is insufficient. + SuspendSettled bool +} + +// ClosesRestoreAttempt validates a never-executed closure against the exact +// persisted request. Absence without this operation-bound evidence is unknown. +func (s ComputeState) ClosesRestoreAttempt(q ResumeRequest) bool { + return q.ReconcileOnly && q.OperationID != "" && s.RestoreAttemptClosed == q.OperationID && s.Status == "absent" && + s.Compute.ID == q.Target.ID && s.Compute.Name == q.Target.Name && s.Compute.Generation == q.Target.Generation && + s.Compute.RestoredFrom != nil && q.Target.RestoredFrom != nil && *s.Compute.RestoredFrom == q.Retained && *q.Target.RestoredFrom == q.Retained && + !s.BootstrapComplete && !s.ResourcesReleased && s.Retained == nil } + type SuspendRequest struct { Reference Reference OperationID string @@ -204,10 +235,10 @@ func ValidateSuspendResult(q SuspendRequest, s ComputeState) error { if ValidateComputeResult(q.Source, s.Compute) != nil { return ErrOwnership } - if !s.SuspendSettled || !s.BootstrapComplete { - return ErrComputeUnconfirmed - } if s.Retained == nil { + if !s.SuspendSettled || !s.BootstrapComplete { + return ErrComputeUnconfirmed + } if !q.ReconcileOnly || s.ResourcesReleased || (s.Status != "running" && s.Status != "paused") { return ErrComputeUnconfirmed } @@ -217,7 +248,7 @@ func ValidateSuspendResult(q SuspendRequest, s ComputeState) error { if ValidateRetained(*v) != nil || v.OperationID != q.OperationID || v.SourceID != q.Source.ID || v.SourceName != q.Source.Name || v.SourceGeneration != q.Source.Generation || (q.Retained != nil && *q.Retained != *v) { return ErrOwnership } - if !s.ResourcesReleased || s.Status != "suspended" { + if !s.SuspendSettled || !s.BootstrapComplete || !s.ResourcesReleased || s.Status != "suspended" { return ErrComputeUnconfirmed } return nil @@ -382,7 +413,7 @@ type Built struct { SpecificationDigest string Provider SandboxProvider InstallationID, BackendFingerprint string - Probe func(context.Context) error + Probe func(context.Context) (*CheckpointCompatibility, error) // Quiescent is nil when no helper can outlive its caller. Quiescent func() bool } diff --git a/services/core/internal/sessions/artifacts_test.go b/services/core/internal/sessions/artifacts_test.go index f11756353..c4213ee9d 100644 --- a/services/core/internal/sessions/artifacts_test.go +++ b/services/core/internal/sessions/artifacts_test.go @@ -12,6 +12,7 @@ import ( "time" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/identity" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/runtimedevice" ) // exportFile is one entry of a test export. @@ -212,7 +213,7 @@ func (s *fakeArtifactStorage) TouchDevice(context.Context, string) (bool, error) return false, nil } -func (s *fakeArtifactStorage) TouchAuthenticatedDevice(context.Context, string, string) (bool, error) { +func (s *fakeArtifactStorage) TouchAuthenticatedDevice(context.Context, string, string, []runtimedevice.SupportedAgentKind) (bool, error) { s.t.Fatal("unexpected call to TouchAuthenticatedDevice") return false, nil } diff --git a/services/core/internal/sessions/creation.go b/services/core/internal/sessions/creation.go index effe001a4..56957dad3 100644 --- a/services/core/internal/sessions/creation.go +++ b/services/core/internal/sessions/creation.go @@ -18,6 +18,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/identity" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/jsonobject" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/metadata" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/skills" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/writeaudit" "github.com/google/uuid" @@ -73,8 +74,6 @@ type CreationTx interface { // LockDeployment locks and reads the sandbox deployment, serializing // hosted creation with deployment changes, restore and node removal. LockDeployment(ctx context.Context) (placement.Deployment, error) - LoadNodes(ctx context.Context) ([]placement.Node, error) - ReservePlacement(ctx context.Context, chosen placement.Placement) error // LockSkills locks the tenant's Skills in ID order, whatever the order of // ids, and returns them by ID. A missing one is ErrNotFound. LockSkills(ctx context.Context, ids []string) (map[string]skills.Skill, error) @@ -114,7 +113,7 @@ type CreationTx interface { // CreateSession creates a Session under a tenant-scoped key, so retries, // including concurrent ones, return the stored Session. The same key with // different input or another creator is ErrIdempotencyConflict. Only the new -// Session freezes resources, takes a placement and admits its initial inputs, +// Session freezes resources and admits its initial inputs, // so a retry after completion or later Turns never submits them again. func (s *Service) CreateSession(ctx context.Context, tenant string, input CreateSession) (Creation, error) { session, batch, encoded, err := prepareCreation(input, s.storage.FingerprintProviderKey) @@ -216,19 +215,18 @@ func (s *Service) createResources(ctx context.Context, tx CreationTx, session Se created = append(created, writeaudit.Resource{Type: writeaudit.ResourceEnvironment, ID: environment, ParentID: session.ID}) } if hosted { - nodes, err := tx.LoadNodes(ctx) - if err != nil { + if err := s.rules.CheckPublicOrigin(deployment.Provider); err != nil { return nil, err } - chosen, err := s.rules.DecidePlacement(deployment, nodes) - if err != nil { + var target sandbox.DeploymentSpec + if err := json.Unmarshal(deployment.Specification, &target); err != nil { return nil, err } - if chosen != nil { - if err := tx.ReservePlacement(ctx, *chosen); err != nil { - return nil, err - } + if err := ValidateRetainedHistory(target.Workspace != nil, input.SupportsRetainedNativeHistory); err != nil { + return nil, err } + // Node placement belongs to the common scheduler, including when + // capacity is available, so new Sessions cannot bypass waiting work. } if len(batch) == 0 { return created, nil diff --git a/services/core/internal/sessions/creation_test.go b/services/core/internal/sessions/creation_test.go index 16f7359ba..7cbc81e34 100644 --- a/services/core/internal/sessions/creation_test.go +++ b/services/core/internal/sessions/creation_test.go @@ -27,8 +27,6 @@ type fakeCreationTx struct { upsertSession func(NewSession) (Creation, error) lockDeployment func() (placement.Deployment, error) - loadNodes func() ([]placement.Node, error) - reservePlacement func() error lockSkills func() (map[string]skills.Skill, error) readSkillVersion func() (skills.Content, error) saveModelExecution func() error @@ -53,16 +51,6 @@ func (f *fakeCreationTx) LockDeployment(context.Context) (placement.Deployment, return f.lockDeployment() } -func (f *fakeCreationTx) LoadNodes(context.Context) ([]placement.Node, error) { - f.record("LoadNodes", f.loadNodes != nil) - return f.loadNodes() -} - -func (f *fakeCreationTx) ReservePlacement(_ context.Context, chosen placement.Placement) error { - f.record("ReservePlacement", f.reservePlacement != nil, chosen.NodeID, fmt.Sprint(chosen.Generation)) - return f.reservePlacement() -} - func (f *fakeCreationTx) LockSkills(_ context.Context, ids []string) (map[string]skills.Skill, error) { f.record("LockSkills", f.lockSkills != nil, ids...) return f.lockSkills() @@ -145,9 +133,9 @@ var errFingerprint = errors.New("fingerprint failed") func failing(string) (string, error) { return "", errFingerprint } // declarations accept every provider and specification. -type declarations struct{} +type declarations struct{ publicOriginRequired bool } -func (declarations) RequiresPublicOrigin(string) (bool, error) { return false, nil } +func (d declarations) RequiresPublicOrigin(string) (bool, error) { return d.publicOriginRequired, nil } func (declarations) ValidateSpecification(string, sandbox.DeploymentSpec) error { return nil } @@ -445,19 +433,21 @@ func TestCreateSession(t *testing.T) { } }) - t.Run("a hosted Session is admitted under the deployment lock and placed after its Environment", func(t *testing.T) { - ready := uint64(5) - tx := newCreationTx(t, hostedSession, true) - tx.lockDeployment = returns(placement.Deployment{InstallationID: "installation", Provider: "docker", Specification: json.RawMessage(`{}`)}) - tx.createEnvironment = returns("environment") - tx.loadNodes = returns([]placement.Node{{ID: "node", Online: true, ServingReady: true, ReadyGeneration: &ready, MaxActive: 1, MaxRetained: 1, CoreURL: rules.PublicURL()}}) - tx.reservePlacement = done - _, err, calls := runCreation(t, rules, tx, creationInput("openai_hosted")) - want := []string{"UpsertSession create", "LockDeployment", "CreateEnvironment", "LoadNodes", "ReservePlacement node 5", "AuditCreation session:session environment:environment:session", "LoadSession"} - if err != nil || strings.Join(calls, "\n") != strings.Join(want, "\n") { - t.Fatalf("calls %q, %v", calls, err) - } - }) + for _, external := range []bool{false, true} { + t.Run(fmt.Sprintf("admission validates target generation external=%t", external), func(t *testing.T) { + tx := newCreationTx(t, hostedSession, true) + spec := json.RawMessage(`{}`) + if external { + spec = json.RawMessage(`{"workspace":{}}`) + } + tx.lockDeployment = returns(placement.Deployment{InstallationID: "installation", Provider: "docker", Mode: string(sandbox.DeploymentNodes), Generation: 5, Specification: spec}) + tx.createEnvironment = returns("environment") + _, err, _ := runCreation(t, rules, tx, creationInput("openai_hosted")) + if external != errors.Is(err, ErrInvalidInput) { + t.Fatalf("target generation admission: %v", err) + } + }) + } for _, test := range []struct { name string @@ -469,10 +459,12 @@ func TestCreateSession(t *testing.T) { {"a reset closes hosted admission", rules, func(tx *fakeCreationTx) { tx.lockDeployment = returns(placement.Deployment{InstallationID: "installation", Provider: "docker", Resetting: true}) }, placement.ErrResetAdmission, []string{"UpsertSession create", "LockDeployment"}}, - {"no node places the Environment", rules, func(tx *fakeCreationTx) { - tx.lockDeployment = returns(placement.Deployment{InstallationID: "installation", Provider: "docker", Specification: json.RawMessage(`{}`)}) - tx.createEnvironment, tx.loadNodes = returns("environment"), returns([]placement.Node(nil)) - }, placement.ErrNodeUnavailable, []string{"UpsertSession create", "LockDeployment", "CreateEnvironment", "LoadNodes"}}, + {"an unconfigured deployment still rejects admission", rules, func(tx *fakeCreationTx) { + tx.lockDeployment = returns(placement.Deployment{Mode: string(sandbox.DeploymentNodes)}) + }, placement.ErrNodeUnavailable, []string{"UpsertSession create", "LockDeployment"}}, + {"an invalid deployment still rejects admission", rules, func(tx *fakeCreationTx) { + tx.lockDeployment = returns(placement.Deployment{InstallationID: "installation", Provider: "docker", Mode: string(sandbox.DeploymentNodes), Specification: json.RawMessage(`[]`)}) + }, placement.ErrAdmissionClosed, []string{"UpsertSession create", "LockDeployment"}}, {"hosted creation needs rules", nil, func(*fakeCreationTx) {}, nil, []string{"UpsertSession create"}}, } { t.Run(test.name, func(t *testing.T) { @@ -517,6 +509,67 @@ func TestCreateSession(t *testing.T) { }) } +func TestCreateSessionDefersNodePlacement(t *testing.T) { + rules, err := placement.NewRules(declarations{}, "https://core.example") + if err != nil { + t.Fatal(err) + } + for _, mode := range []sandbox.DeploymentMode{sandbox.DeploymentNodes, sandbox.DeploymentDirect} { + for _, initialInput := range []bool{false, true} { + t.Run(fmt.Sprintf("%s/initial=%t", mode, initialInput), func(t *testing.T) { + input := creationInput("openai_hosted") + input.SupportsRetainedNativeHistory = true + if initialInput { + input.InitialInputs = []Input{messageInput("hi")} + } + session := Session{ID: "session", Engine: "codex", Configuration: input.Configuration} + tx := newCreationTx(t, session, true) + tx.lockDeployment = returns(placement.Deployment{InstallationID: "installation", Provider: "fixture", Mode: string(mode), Generation: 5, Specification: json.RawMessage(`{"workspace":{}}`)}) + tx.createEnvironment = returns("environment") + if initialInput { + tx.loadEnvironmentInput, tx.createInputReservation, tx.pruneChanges = inputs(), returns(EnvironmentInputReservation{}), done + } + result, err, calls := runCreation(t, rules, tx, input) + if err != nil || !result.Created { + t.Fatalf("creation %+v, %v", result, err) + } + want := []string{"UpsertSession create", "LockDeployment", "CreateEnvironment"} + if initialInput { + want = append(want, "LoadEnvironmentInput", `CreateInputReservation [{"kind":"message","payload":`+hi+`}] initial`, "LoadEnvironmentInput", "PruneChanges") + } + want = append(want, "AuditCreation session:session environment:environment:session", "LoadSession") + if strings.Join(calls, "\n") != strings.Join(want, "\n") { + t.Fatalf("calls %q, want %q", calls, want) + } + // A retry reads the durable Session without reserving input again. + retry := newCreationTx(t, session, false) + result, err, calls = runCreation(t, rules, retry, input) + if err != nil || result.Created || result.Session.ID != session.ID || strings.Join(calls, "\n") != "UpsertSession create\nAuditCreation\nLoadSession" { + t.Fatalf("retry %+v, calls %q, %v", result, calls, err) + } + }) + } + } +} + +func TestCreateSessionStillChecksPublicOriginBeforePlacement(t *testing.T) { + rules, err := placement.NewRules(declarations{publicOriginRequired: true}, "http://localhost:8080") + if err != nil { + t.Fatal(err) + } + for _, mode := range []sandbox.DeploymentMode{sandbox.DeploymentNodes, sandbox.DeploymentDirect} { + t.Run(string(mode), func(t *testing.T) { + input := creationInput("openai_hosted") + tx := newCreationTx(t, Session{ID: "session", Configuration: input.Configuration}, true) + tx.lockDeployment = returns(placement.Deployment{InstallationID: "installation", Provider: "fixture", Mode: string(mode), Specification: json.RawMessage(`{}`)}) + tx.createEnvironment = returns("environment") + if _, err, _ := runCreation(t, rules, tx, input); !errors.Is(err, placement.ErrPublicURLUnreachable) { + t.Fatalf("unreachable public origin admitted: %v", err) + } + }) + } +} + func TestFindSessionCreation(t *testing.T) { request := json.RawMessage(`{"agent_id":"agent"}`) hash, err := intentHash(request, fingerprints) diff --git a/services/core/internal/sessions/devices.go b/services/core/internal/sessions/devices.go index d7fd40cf9..1a488b93c 100644 --- a/services/core/internal/sessions/devices.go +++ b/services/core/internal/sessions/devices.go @@ -94,7 +94,7 @@ type DeviceStorage interface { TouchDevice(ctx context.Context, device string) (bool, error) // TouchAuthenticatedDevice records that the device was seen with the // credential and reports whether that credential still has authority. - TouchAuthenticatedDevice(ctx context.Context, device, credentialHash string) (bool, error) + TouchAuthenticatedDevice(ctx context.Context, device, credentialHash string, kinds []runtimedevice.SupportedAgentKind) (bool, error) // WithEnrollment authenticates the executor credential for the // Environment, then runs apply in the transaction of the Environment's // Session, with the Environment and what the Session lock shows. A @@ -148,7 +148,7 @@ func (s *Service) TouchRuntimeHeartbeat(ctx context.Context, device string) (run // credential the gateway authenticated. A credential that lost its authority // reports Deleted. func (s *Service) TouchAgentDaemonHeartbeat(ctx context.Context, heartbeat runtimedevice.Heartbeat) (runtimedevice.HeartbeatStatus, error) { - current, err := s.storage.TouchAuthenticatedDevice(ctx, heartbeat.RuntimeID, heartbeat.CredentialHash) + current, err := s.storage.TouchAuthenticatedDevice(ctx, heartbeat.RuntimeID, heartbeat.CredentialHash, heartbeat.SupportedAgentKinds) if err != nil { return runtimedevice.HeartbeatStatus{}, err } @@ -156,7 +156,8 @@ func (s *Service) TouchAgentDaemonHeartbeat(ctx context.Context, heartbeat runti } // heartbeatStatus is the liveness a heartbeat reports. Live connectivity -// belongs to the gateway Registry; only the last-seen time is stored. +// belongs to the gateway Registry. Stored declarations describe capabilities, +// not current connectivity. func heartbeatStatus(current bool) runtimedevice.HeartbeatStatus { return runtimedevice.HeartbeatStatus{Liveness: "online", Deleted: !current} } diff --git a/services/core/internal/sessions/devices_test.go b/services/core/internal/sessions/devices_test.go index 0ea72fcd4..6fa241645 100644 --- a/services/core/internal/sessions/devices_test.go +++ b/services/core/internal/sessions/devices_test.go @@ -14,8 +14,9 @@ import ( // identify it, then runs its func; a method whose func is unset fails the // test. type fakeStorage struct { - t *testing.T - calls []string + t *testing.T + calls []string + heartbeatKinds []runtimedevice.SupportedAgentKind createDevice func(DeviceRegistration) (ExecutionDevice, error) revokeDevice func() error @@ -76,8 +77,9 @@ func (s *fakeStorage) TouchDevice(_ context.Context, device string) (bool, error return s.touchDevice() } -func (s *fakeStorage) TouchAuthenticatedDevice(_ context.Context, device, credentialHash string) (bool, error) { +func (s *fakeStorage) TouchAuthenticatedDevice(_ context.Context, device, credentialHash string, kinds []runtimedevice.SupportedAgentKind) (bool, error) { s.record("TouchAuthenticatedDevice", s.touchAuthenticatedDevice != nil, device, credentialHash) + s.heartbeatKinds = kinds return s.touchAuthenticatedDevice() } @@ -233,3 +235,16 @@ func TestEnrollRuntimeBindsUnderTheSessionLock(t *testing.T) { }) } } + +func TestHeartbeatPassesRetainedHistoryDeclarationToAuthenticatedStorage(t *testing.T) { + storage := &fakeStorage{t: t, touchAuthenticatedDevice: returns(true)} + kinds := []runtimedevice.SupportedAgentKind{{Kind: "fixture", Available: true, Capabilities: runtimedevice.KindCapabilities{RetainedNativeHistory: true}}} + _, err := deviceService(t, storage).TouchAgentDaemonHeartbeat(t.Context(), runtimedevice.Heartbeat{RuntimeID: "device", CredentialHash: credentialDigest, SupportedAgentKinds: kinds}) + if err != nil || len(storage.heartbeatKinds) != 1 || !storage.heartbeatKinds[0].Capabilities.RetainedNativeHistory { + t.Fatalf("declaration lost: %v", err) + } + _, err = deviceService(t, storage).TouchAgentDaemonHeartbeat(t.Context(), runtimedevice.Heartbeat{RuntimeID: "device", CredentialHash: credentialDigest}) + if err != nil || len(storage.heartbeatKinds) != 0 { + t.Fatalf("removed declaration retained: %v", err) + } +} diff --git a/services/core/internal/sessions/environment.go b/services/core/internal/sessions/environment.go index 2a11a64b8..dd0afb7ae 100644 --- a/services/core/internal/sessions/environment.go +++ b/services/core/internal/sessions/environment.go @@ -61,13 +61,15 @@ func createsEnvironment(configuration json.RawMessage) (bool, error) { // Environment retains execution ownership; its configuration is an internal snapshot, not a public response. type Environment struct { - Initialization string - ID string - SessionID string - TenantID string - Status string - CreatedAt time.Time - Configuration json.RawMessage + // ExternalWorkspace is a read projection of the immutable filesystem binding, not a setting. + ExternalWorkspace bool + Initialization string + ID string + SessionID string + TenantID string + Status string + CreatedAt time.Time + Configuration json.RawMessage } // EnvironmentInitialization owns preparation independently of compute ownership. @@ -135,3 +137,11 @@ type EnvironmentFileWrite struct { SettledAt *time.Time Replayed bool } + +// ValidateRetainedHistory checks the selected external filesystem and Harness combination. +func ValidateRetainedHistory(required, supported bool) error { + if required && !supported { + return fmt.Errorf("%w: external workspace storage requires retained_native_history", ErrInvalidInput) + } + return nil +} diff --git a/services/core/internal/sessions/session.go b/services/core/internal/sessions/session.go index 9ce349aa4..534ac908d 100644 --- a/services/core/internal/sessions/session.go +++ b/services/core/internal/sessions/session.go @@ -42,6 +42,9 @@ type Session struct { } type CreateSession struct { + // SupportsRetainedNativeHistory is trusted admission input from the immutable engine profile. + // It is not persisted and never contributes to the caller's retry identity. + SupportsRetainedNativeHistory bool `json:"-"` // DeploymentProviderRevision is private creation metadata, never retry identity. DeploymentProviderRevision uuid.UUID `json:"-"` ExecutionConfiguration *v1.SessionExecutionConfiguration diff --git a/services/core/internal/sessions/transaction.go b/services/core/internal/sessions/transaction.go index 10e58f34c..6d5b84542 100644 --- a/services/core/internal/sessions/transaction.go +++ b/services/core/internal/sessions/transaction.go @@ -14,6 +14,8 @@ import ( // LockedSession is what locking a Session row shows. type LockedSession struct { + // Engine is the immutable Harness selected when the Session was created. + Engine string // Deleted reports that the Session was publicly deleted. Its row stays so // that remaining execution can settle. Deleted bool diff --git a/services/core/migrations/000098_compute_incarnations.sql b/services/core/migrations/000098_compute_incarnations.sql new file mode 100644 index 000000000..7fb951331 --- /dev/null +++ b/services/core/migrations/000098_compute_incarnations.sql @@ -0,0 +1,40 @@ +-- +goose Up +ALTER TABLE runtime_allocations DROP CONSTRAINT runtime_allocations_environment_id_key; +CREATE UNIQUE INDEX runtime_allocations_current_environment ON runtime_allocations(environment_id) WHERE state <> 'released'; +CREATE INDEX runtime_allocations_environment_receipts ON runtime_allocations(environment_id, created_at DESC, id DESC); +ALTER TABLE runtime_allocations DROP CONSTRAINT runtime_allocation_placement_fk; +ALTER TABLE runtime_allocations ADD CONSTRAINT runtime_allocation_node_fk FOREIGN KEY(node_id) REFERENCES runtime_nodes(id); +ALTER TABLE devices DROP CONSTRAINT devices_environment_id_key; +CREATE UNIQUE INDEX devices_environment_authority ON devices(environment_id) WHERE revoked_at IS NULL OR executor_key_id IS NOT NULL; + +ALTER TABLE devices ADD COLUMN supported_agent_kinds jsonb NOT NULL DEFAULT '[]' CHECK (jsonb_typeof(supported_agent_kinds) = 'array'); + +-- Cold retained Sessions own no compute, but a reset must end their execution lifetime. +CREATE VIEW runtime_reset_retained_environments AS +SELECT e.id AS environment_id, s.id AS session_id, s.tenant_id, a.deployment_generation +FROM environments e JOIN sessions s ON s.id = e.session_id +JOIN environment_workspaces w ON w.environment_id = e.id AND w.state = 'ready' +JOIN LATERAL ( + SELECT allocation.* FROM runtime_allocations allocation + WHERE allocation.environment_id = e.id + ORDER BY allocation.created_at DESC, allocation.id DESC LIMIT 1 +) a ON true +CROSS JOIN runtime_deployment d +WHERE d.reset_clear IS NOT NULL AND d.reset_requested_at IS NOT NULL + AND s.deleted_at IS NULL AND s.created_at <= d.reset_requested_at + AND e.status NOT IN ('failed','expired') AND e.initialization = 'complete' + AND s.configuration->'environment'->>'type' = 'openai_hosted' + AND a.state = 'released' AND a.provider_key = d.installation_id + AND a.deployment_generation <= d.generation AND a.created_at <= d.reset_requested_at + AND NOT EXISTS (SELECT 1 FROM runtime_placements p WHERE p.environment_id = e.id AND p.released_at IS NULL); + +-- +goose Down +DROP VIEW runtime_reset_retained_environments; +ALTER TABLE devices DROP COLUMN supported_agent_kinds; +DROP INDEX devices_environment_authority; +ALTER TABLE devices ADD CONSTRAINT devices_environment_id_key UNIQUE(environment_id); +ALTER TABLE runtime_allocations DROP CONSTRAINT runtime_allocation_node_fk; +ALTER TABLE runtime_allocations ADD CONSTRAINT runtime_allocation_placement_fk FOREIGN KEY(environment_id,node_id) REFERENCES runtime_placements(environment_id,node_id); +DROP INDEX runtime_allocations_environment_receipts; +DROP INDEX runtime_allocations_current_environment; +ALTER TABLE runtime_allocations ADD CONSTRAINT runtime_allocations_environment_id_key UNIQUE(environment_id); diff --git a/services/core/migrations/000099_checkpoint_compatibility.sql b/services/core/migrations/000099_checkpoint_compatibility.sql new file mode 100644 index 000000000..0e9b0b490 --- /dev/null +++ b/services/core/migrations/000099_checkpoint_compatibility.sql @@ -0,0 +1,6 @@ +-- +goose Up +ALTER TABLE runtime_node_generation_status ADD COLUMN checkpoint jsonb + CHECK (checkpoint IS NULL OR jsonb_typeof(checkpoint) = 'object'); + +-- +goose Down +ALTER TABLE runtime_node_generation_status DROP COLUMN checkpoint; diff --git a/services/core/migrations/beta_upgrade_test.go b/services/core/migrations/beta_upgrade_test.go index 940ac9901..44a044e57 100644 --- a/services/core/migrations/beta_upgrade_test.go +++ b/services/core/migrations/beta_upgrade_test.go @@ -3,6 +3,7 @@ package migrations import ( "context" "database/sql" + "errors" "fmt" "os" "strings" @@ -10,11 +11,12 @@ import ( "github.com/google/uuid" "github.com/jackc/pgx/v5" -"github.com/jackc/pgx/v5/stdlib" + "github.com/jackc/pgx/v5/pgconn" + "github.com/jackc/pgx/v5/stdlib" "github.com/pressly/goose/v3" ) -// A dedicated disposable database exercises the real 91-to-97 SQL, never a +// A dedicated disposable database exercises the real 91-to-99 SQL, never a // product database. No provider or native sandbox is contacted by this test. func TestBeta91UpgradePreservesRetainedAllocations(t *testing.T) { dsn := os.Getenv("OAC_TEST_DATABASE_URL") @@ -94,7 +96,7 @@ func TestBeta91UpgradePreservesRetainedAllocations(t *testing.T) { t.Fatal("migration changed retained state, ownership or deadline") } var version, live, workspaces int - if err = db.QueryRowContext(t.Context(), `SELECT max(version_id) FROM agents_api_schema_version WHERE is_applied`).Scan(&version); err != nil || version != 97 { + if err = db.QueryRowContext(t.Context(), `SELECT max(version_id) FROM agents_api_schema_version WHERE is_applied`).Scan(&version); err != nil || version != 99 { t.Fatal(version, err) } if err = db.QueryRowContext(t.Context(), `SELECT count(*) FROM runtime_allocations WHERE state<>'released' AND compute_phase='suspended' AND compute_state->>'protocol_version'='1'`).Scan(&live); err != nil || live != 17 { @@ -103,6 +105,11 @@ func TestBeta91UpgradePreservesRetainedAllocations(t *testing.T) { if err = db.QueryRowContext(t.Context(), `SELECT count(*) FROM environment_workspaces`).Scan(&workspaces); err != nil || workspaces != 0 { t.Fatal(workspaces, err) } + checkBeta99CheckpointConstraint(t, db, installation) + checkBeta98HistoricalAllocationDowngrade(t, db, provider) + if snapshot() != before { + t.Fatal("constraint checks changed retained ownership") + } // Empty workspace ownership permits Down, but it does not restore removed // values. Verify the documented E2B-policy rollback incompatibility. if _, err = provider.DownTo(t.Context(), 91); err != nil { @@ -116,3 +123,87 @@ func TestBeta91UpgradePreservesRetainedAllocations(t *testing.T) { t.Fatal("down migration changed retained ownership") } } + +// Migration 99 accepts only structured compatibility evidence. It does not +// manufacture a token for existing nodes or reinterpret adapter-private state. +func checkBeta99CheckpointConstraint(t *testing.T, db *sql.DB, installation string) { + t.Helper() + node, connection := uuid.NewString(), uuid.NewString() + if _, err := db.ExecContext(t.Context(), `INSERT INTO runtime_nodes(id,installation_id,name,backend_fingerprint,credential_sha256,max_active,max_retained) VALUES($1,$2,'migration-fixture',repeat('c',64),repeat('d',64),1,2)`, node, installation); err != nil { + t.Fatal(err) + } + if _, err := db.ExecContext(t.Context(), `INSERT INTO runtime_node_generation_status(node_id,generation,specification_digest,connection_id,owner_epoch,state) VALUES($1,1,repeat('e',64),$2,1,'ready')`, node, connection); err != nil { + t.Fatal(err) + } + var absent bool + if err := db.QueryRowContext(t.Context(), `SELECT checkpoint IS NULL FROM runtime_node_generation_status WHERE node_id=$1`, node).Scan(&absent); err != nil || !absent { + t.Fatal("checkpoint default", absent, err) + } + if _, err := db.ExecContext(t.Context(), `UPDATE runtime_node_generation_status SET checkpoint='{"artifact_domain":"fixture","execution_class":"fixture"}' WHERE node_id=$1`, node); err != nil { + t.Fatal(err) + } + for _, invalid := range []string{`[]`, `"opaque"`, `1`, `null`} { + _, err := db.ExecContext(t.Context(), `UPDATE runtime_node_generation_status SET checkpoint=$2 WHERE node_id=$1`, node, invalid) + var pgerr *pgconn.PgError + if !errors.As(err, &pgerr) || pgerr.Code != "23514" { + t.Fatalf("checkpoint shape %s: %v", invalid, err) + } + } + if _, err := db.ExecContext(t.Context(), `DELETE FROM runtime_node_generation_status WHERE node_id=$1`, node); err != nil { + t.Fatal(err) + } + if _, err := db.ExecContext(t.Context(), `DELETE FROM runtime_nodes WHERE id=$1`, node); err != nil { + t.Fatal(err) + } +} + +// Migration 98 keeps multiple historical receipts but one active allocation +// and one live device authority. Down must fail rather than erase history. +func checkBeta98HistoricalAllocationDowngrade(t *testing.T, db *sql.DB, provider *goose.Provider) { + t.Helper() + device, allocation := uuid.NewString(), uuid.NewString() + var environment, tenant, installation string + if err := db.QueryRowContext(t.Context(), `SELECT a.environment_id,s.tenant_id,a.provider_key FROM runtime_allocations a JOIN environments e ON e.id=a.environment_id JOIN sessions s ON s.id=e.session_id WHERE a.state='running' ORDER BY a.id LIMIT 1`).Scan(&environment, &tenant, &installation); err != nil { + t.Fatal(err) + } + _, err := db.ExecContext(t.Context(), `INSERT INTO devices(id,tenant_id,name,credential_hash,environment_id) VALUES($1,$2,'duplicate-authority',repeat('f',64),$3)`, device, tenant, environment) + requireBetaUniqueViolation(t, err, "devices_environment_authority") + if _, err := db.ExecContext(t.Context(), `INSERT INTO devices(id,tenant_id,name,credential_hash,environment_id,revoked_at) VALUES($1,$2,'historical-device',repeat('f',64),$3,clock_timestamp())`, device, tenant, environment); err != nil { + t.Fatal(err) + } + _, err = db.ExecContext(t.Context(), `INSERT INTO runtime_allocations(id,environment_id,device_id,provider_key,state,create_settled,deployment_generation) VALUES($1,$2,$3,$4,'running',true,4)`, allocation, environment, device, installation) + requireBetaUniqueViolation(t, err, "runtime_allocations_current_environment") + if _, err := db.ExecContext(t.Context(), `INSERT INTO runtime_allocations(id,environment_id,device_id,provider_key,state,create_settled,released_at,deployment_generation) VALUES($1,$2,$3,$4,'released',true,clock_timestamp(),4)`, allocation, environment, device, installation); err != nil { + t.Fatal(err) + } + _, err = provider.DownTo(t.Context(), 97) + requireBetaUniqueViolation(t, err, "devices_environment_id_key") + // Isolate the allocation downgrade guard after proving device history also + // prevents Down. This fixture never grants the unbound old device authority. + if _, err := db.ExecContext(t.Context(), `UPDATE devices SET environment_id=NULL WHERE id=$1`, device); err != nil { + t.Fatal(err) + } + _, err = provider.DownTo(t.Context(), 97) + requireBetaUniqueViolation(t, err, "runtime_allocations_environment_id_key") + var count int + if err := db.QueryRowContext(t.Context(), `SELECT count(*) FROM runtime_allocations WHERE environment_id=$1`, environment).Scan(&count); err != nil || count != 2 { + t.Fatal("failed downgrade lost history", count, err) + } + if _, err := db.ExecContext(t.Context(), `DELETE FROM runtime_allocations WHERE id=$1`, allocation); err != nil { + t.Fatal(err) + } + if _, err := db.ExecContext(t.Context(), `DELETE FROM devices WHERE id=$1`, device); err != nil { + t.Fatal(err) + } + if _, err := provider.Up(t.Context()); err != nil { + t.Fatal(err) + } +} + +func requireBetaUniqueViolation(t *testing.T, err error, constraint string) { + t.Helper() + var pgerr *pgconn.PgError + if !errors.As(err, &pgerr) || pgerr.Code != "23505" || pgerr.ConstraintName != constraint { + t.Fatalf("expected %s uniqueness fence: %v", constraint, err) + } +} diff --git a/services/core/tests/integration/admin_session_archive_test.go b/services/core/tests/integration/admin_session_archive_test.go index 810a5ee48..5101af9b1 100644 --- a/services/core/tests/integration/admin_session_archive_test.go +++ b/services/core/tests/integration/admin_session_archive_test.go @@ -69,6 +69,15 @@ func managedArchiveSession(t *testing.T, s *Store, input sessions.CreateSession) func archiveAllocation(t *testing.T, w *Store, tenant string, session sessions.Session, installation string) deployment.Allocation { t.Helper() + setup, err := deploymentService(t, w).Setup(t.Context()) + if err != nil { + t.Fatal(err) + } + if setup.Mode == "nodes" { + if err := reserveSessionPlacement(t, w, w, session); err != nil { + t.Fatal(err) + } + } owner, err := deploymentExecution(t, w).ReserveAllocation(t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: session.Environment.ID}, installation, runtimedevice.HashCredential(uuid.NewString())) if err != nil { t.Fatal(err) diff --git a/services/core/tests/integration/deployment_fixture_test.go b/services/core/tests/integration/deployment_fixture_test.go index 25018ae00..2073a5b60 100644 --- a/services/core/tests/integration/deployment_fixture_test.go +++ b/services/core/tests/integration/deployment_fixture_test.go @@ -1,12 +1,14 @@ package integration import ( + "context" "testing" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/deploymentpg" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/persistence/postgres/pgunit" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/providers" ) @@ -50,3 +52,16 @@ func deploymentExecution(t testing.TB, w *Store) *deployment.ExecutionOperations } return operations } + +func fixtureNodeHeartbeat(ctx context.Context, store *Store, service *deployment.Service, nodeID, connection string, epoch uint64, health deployment.NodeHealth) error { + var generation uint64 + var digest, provider string + if err := store.pool.QueryRow(ctx, "SELECT n.deployment_generation,n.specification_digest,d.provider_kind FROM runtime_nodes n CROSS JOIN runtime_deployment d WHERE n.id=$1", nodeID).Scan(&generation, &digest, &provider); err != nil { + return err + } + var statuses []sandbox.GenerationStatus + if health.ProviderReady && provider == "microsandbox" { + statuses = []sandbox.GenerationStatus{{Generation: generation, SpecificationDigest: digest, State: "ready", Checkpoint: &sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}}} + } + return service.Heartbeat(ctx, nodeID, connection, epoch, health, statuses) +} diff --git a/services/core/tests/integration/deployment_public_fixture_test.go b/services/core/tests/integration/deployment_public_fixture_test.go index b1ce77186..565cdab05 100644 --- a/services/core/tests/integration/deployment_public_fixture_test.go +++ b/services/core/tests/integration/deployment_public_fixture_test.go @@ -1,6 +1,7 @@ package integration import ( + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/engine" "testing" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" @@ -34,6 +35,6 @@ func fixtureDeploymentExecution(s *Store, lease *pgunit.Lease) (*deployment.Serv if err != nil { return nil, nil, err } - changes, err := deployment.NewExecutionOperations(service, deploymentpg.NewExecution(lease, s.credentialCipher)) + changes, err := deployment.NewExecutionOperations(service, deploymentpg.NewExecution(lease, s.credentialCipher), engine.Catalog{}) return service, changes, err } diff --git a/services/core/tests/integration/device_bootstrap_binding_test.go b/services/core/tests/integration/device_bootstrap_binding_test.go index 4d85bc1c9..f6187833c 100644 --- a/services/core/tests/integration/device_bootstrap_binding_test.go +++ b/services/core/tests/integration/device_bootstrap_binding_test.go @@ -28,7 +28,7 @@ func TestDeviceCredentialCarriesPersistedAllocationNode(t *testing.T) { for _, nodeID := range []string{d.NodeID, remote} { t.Run(nodeID, func(t *testing.T) { tenant, bearer := uuid.NewString(), uuid.NewString() - session, err := createSessionOnNode(t, s, tenant, managerSessionInput(uuid.NewString()), nodeID) + session, err := createSessionOnNode(t, s, writer, tenant, managerSessionInput(uuid.NewString()), nodeID) if err != nil { t.Fatal(err) } diff --git a/services/core/tests/integration/environment_file_write_semantics_public_test.go b/services/core/tests/integration/environment_file_write_semantics_public_test.go index eef998402..c8e66c22e 100644 --- a/services/core/tests/integration/environment_file_write_semantics_public_test.go +++ b/services/core/tests/integration/environment_file_write_semantics_public_test.go @@ -59,7 +59,7 @@ func TestEnvironmentFileCreateRejectionsLeaveNoReceiptOrConsumption(t *testing.T {OrganizationID: "test-org", ProjectID: uuid.NewString(), SubjectKind: "service_account", SubjectID: "test-runner", TokenSHA256: runtimedevice.HashCredential(token), TenantID: h.tenant}, {OrganizationID: "test-org", ProjectID: uuid.NewString(), SubjectKind: "service_account", SubjectID: "tenant-b", TokenSHA256: runtimedevice.HashCredential(other), TenantID: uuid.NewString()}, }) - handler, err := publicHandler(t, h.s, auth, "codex", workerExecution(t, w)) + handler, err := publicHandler(t, h.s, auth, "codex", workerExecution(t, w), acceptUnavailable(t)) if err != nil { t.Fatal(err) } @@ -144,6 +144,16 @@ func TestEnvironmentFileCreateRejectionsLeaveNoReceiptOrConsumption(t *testing.T t.Fatal("rejection did not settle", intent, err) } } + // Exhaustion rejected before publication is unavailable, not invalid input; + // its exact receipt still settles ownership so another write can proceed. + doneUnavailable := post(token, environment.ID, inline) + unavailableID := serveFileWrite(h, proto.WorkspaceWriteResultPayload{Outcome: "rejected", ErrorCode: "resource_unavailable"}) + if got := await(doneUnavailable); got.status != http.StatusServiceUnavailable { + t.Fatal("resource rejection became invalid input", got) + } + if intent, err := FixtureFileWrite(t.Context(), h.s.pool, h.tenant, environment.ID, unavailableID); err != nil || intent.State != "rejected" { + t.Fatal("known unavailable result retained uncertainty", intent, err) + } // The rejected copy did not consume its Source File. if got, err := fileStore.Get(t.Context(), h.tenant, source.ID); err != nil || got.SizeBytes != 3 { t.Fatal("source file consumed", got, err) @@ -156,7 +166,7 @@ func TestEnvironmentFileCreateRejectionsLeaveNoReceiptOrConsumption(t *testing.T if foreign.status != 404 || foreign != missing { t.Fatal("tenant B reached the Environment", foreign, missing) } - if got := states(); !reflect.DeepEqual(got, map[string]int{"rejected": 3}) { + if got := states(); !reflect.DeepEqual(got, map[string]int{"rejected": 4}) { t.Fatal("rejections left a receipt or blocking intent", got) } // Known rejections release the mutation owner for a successor. @@ -165,7 +175,7 @@ func TestEnvironmentFileCreateRejectionsLeaveNoReceiptOrConsumption(t *testing.T if got := await(done); got.status != 201 { t.Fatal("successor rejected", got) } - if got := states(); !reflect.DeepEqual(got, map[string]int{"rejected": 3, "committed": 1}) { + if got := states(); !reflect.DeepEqual(got, map[string]int{"rejected": 4, "committed": 1}) { t.Fatal("successor receipt", got) } session, err := sessionAdapter(h.s).GetSession(t.Context(), h.tenant, h.session.ID) diff --git a/services/core/tests/integration/execution_test.go b/services/core/tests/integration/execution_test.go index 8e73078db..66a31026c 100644 --- a/services/core/tests/integration/execution_test.go +++ b/services/core/tests/integration/execution_test.go @@ -99,7 +99,8 @@ func executionOwnerPID(t *testing.T, pool *pgxpool.Pool) int32 { t.Helper() var pid int32 err := pool.QueryRow(t.Context(), `SELECT pid FROM pg_locks WHERE locktype='advisory' AND granted AND objsubid=1 - AND classid::bigint * 4294967296 + objid::bigint = 706172736172 + AND classid::bigint = (706172736172::bigint >> 32) + AND objid::bigint = (706172736172::bigint & 4294967295) AND database=(SELECT oid FROM pg_database WHERE datname=current_database())`).Scan(&pid) if err != nil { t.Fatal("observe execution lease owner", err) diff --git a/services/core/tests/integration/runtime_allocations_test.go b/services/core/tests/integration/runtime_allocations_test.go index cd9899680..824ab501a 100644 --- a/services/core/tests/integration/runtime_allocations_test.go +++ b/services/core/tests/integration/runtime_allocations_test.go @@ -183,8 +183,12 @@ func TestRuntimeAllocationCleanupRevokesAndKeepsIdentity(t *testing.T) { if _, err := deploymentExecution(t, w).ReleaseAllocation(t.Context(), owner); err != nil { t.Fatal(err) } - got, err := deploymentExecution(t, w).ReserveAllocation(t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: environment.ID}, owner.ProviderKey, runtimedevice.HashCredential(uuid.NewString())) - if err != nil || !got.Replayed || got.State != "released" { - t.Fatalf("cleanup permitted replacement: %+v %v", got, err) + key := deployment.AllocationKey{TenantID: tenant, EnvironmentID: environment.ID} + if _, err := deploymentExecution(t, w).ReserveAllocation(t.Context(), key, owner.ProviderKey, runtimedevice.HashCredential(uuid.NewString())); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatalf("terminal cleanup permitted replacement: %v", err) + } + got, err := deploymentStore(s).EnvironmentAllocation(t.Context(), key) + if err != nil || got.ID != owner.ID || got.DeviceID != owner.DeviceID || got.State != "released" { + t.Fatalf("cleanup changed retained ownership: %+v %v", got, err) } } diff --git a/services/core/tests/integration/runtime_compute_lifecycle_test.go b/services/core/tests/integration/runtime_compute_lifecycle_test.go index 53f678b7d..e37f43ba7 100644 --- a/services/core/tests/integration/runtime_compute_lifecycle_test.go +++ b/services/core/tests/integration/runtime_compute_lifecycle_test.go @@ -15,6 +15,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto" "github.com/MiniMax-AI/OpenAgentCore/internal/agentdaemon/proto/prototest" + db "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/db/sqlc" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/execution" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/runtimegateway" @@ -93,7 +94,7 @@ func (p *fakeSuspensionProvider) Suspend(_ context.Context, q sandbox.SuspendReq if _, exists := p.snapshots[q.OperationID]; exists { return sandbox.ComputeState{}, errors.New("capture replayed") } - p.snapshots[q.OperationID] = sandbox.RetainedState{Reference: "snapshot-" + q.OperationID, ID: uuid.NewString(), Data: "verified-native-state", OperationID: q.OperationID, SourceGeneration: q.Source.Generation, SourceName: q.Source.Name, SourceID: q.Source.ID} + p.snapshots[q.OperationID] = sandbox.RetainedState{Compatibility: sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, Reference: "snapshot-" + q.OperationID, ID: uuid.NewString(), Data: "verified-native-state", OperationID: q.OperationID, SourceGeneration: q.Source.Generation, SourceName: q.Source.Name, SourceID: q.Source.ID} state.Status = "paused" p.computes[q.Source.Name] = state if p.loseCapture { @@ -371,7 +372,7 @@ func (f *computeLifecycleFixture) phase(tenant, environment, phase string) deplo func (f *computeLifecycleFixture) complete(owner deployment.Allocation) string { id := uuid.NewString() f.sql(`INSERT INTO turns(id,session_id,status,completed_at) VALUES($1,$2,'completed',clock_timestamp()-interval '2 minutes')`, id, owner.SessionID) - f.sql(`UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '2 minutes' WHERE id=$1`, owner.ID) + f.sql(`UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '2 minutes',compute_phase_changed_at=clock_timestamp()-interval '2 minutes' WHERE id=$1`, owner.ID) return id } func (f *computeLifecycleFixture) queued(owner deployment.Allocation) string { @@ -414,7 +415,7 @@ func TestRuntimeComputeLifecycleIdleSuspendAndQueuedSameSessionWake(t *testing.T func TestRuntimeComputeLifecycleUnusedSessionSuspendsAndExpires(t *testing.T) { f := newComputeLifecycleFixture(t, 1, 2) tenant, session, env, owner := f.create() - f.sql(`UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '2 minutes' WHERE id=$1`, owner.ID) + f.sql(`UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '2 minutes',compute_phase_changed_at=clock_timestamp()-interval '2 minutes' WHERE id=$1`, owner.ID) suspended := f.phase(tenant, env.ID, "suspended") if suspended.ComputeRetainedUntil == nil || f.provider.captures != 1 || f.provider.computeKills != 1 || len(f.provider.computes) != 0 || len(f.provider.snapshots) != 1 { t.Fatal("unused Session did not suspend and release active compute") @@ -614,21 +615,41 @@ func TestRuntimeComputeProtocolUpgradeRefusesOldReceiptsBeforeCleanup(t *testing f.complete(owner) retained := f.phase(tenant, env.ID, "suspended") f.stop() - f.sql(`UPDATE runtime_allocations SET compute_state=(compute_state-'protocol_version'-'retained') || jsonb_build_object('snapshot',compute_state->'retained') WHERE id=$1`, owner.ID) - deletes := f.provider.snapshotDeletes - w, err := startNextWorker(t, t.Context(), f.store, &execution.Dispatcher{Registry: f.provider.registry, ManagedRuntimes: webRuntimes(t, f.store, f.key, f.provider, &f.policy)}) - if err == nil || w != nil || !strings.Contains(err.Error(), "previous release") { - t.Fatalf("incompatible state activated: %v", err) - } - if f.provider.snapshotDeletes != deletes || len(f.provider.snapshots) != 1 { - t.Fatal("upgrade lost owned artifact") + for _, versioned := range []bool{false, true} { + f.sql(`UPDATE runtime_allocations SET compute_state=($2::jsonb-'retained') || jsonb_build_object('snapshot',$2::jsonb->'retained') WHERE id=$1`, owner.ID, retained.ComputeState) + if !versioned { + f.sql(`UPDATE runtime_allocations SET compute_state=compute_state-'protocol_version' WHERE id=$1`, owner.ID) + } + deletes := f.provider.snapshotDeletes + w, err := startNextWorker(t, t.Context(), f.store, &execution.Dispatcher{Registry: f.provider.registry, ManagedRuntimes: webRuntimes(t, f.store, f.key, f.provider, &f.policy)}) + if err == nil || w != nil || !strings.Contains(err.Error(), "previous release") { + t.Fatalf("incompatible state activated: %v", err) + } + if f.provider.snapshotDeletes != deletes || len(f.provider.snapshots) != 1 { + t.Fatal("upgrade lost owned artifact") + } + var state []byte + if err := f.store.pool.QueryRow(t.Context(), `SELECT compute_state FROM runtime_allocations WHERE id=$1`, owner.ID).Scan(&state); err != nil { + t.Fatal(err) + } + if !strings.Contains(string(state), `"snapshot"`) { + t.Fatal("old receipt was rewritten") + } } - var state []byte - if err := f.store.pool.QueryRow(t.Context(), `SELECT compute_state FROM runtime_allocations WHERE id=$1`, owner.ID).Scan(&state); err != nil { - t.Fatal(err) + // A consumed legacy snapshot can survive only in current/target provenance. + for _, field := range []string{"current", "target"} { + f.sql(`UPDATE runtime_allocations SET compute_state=$2::jsonb || jsonb_build_object($3::text, jsonb_build_object('RestoredFrom', jsonb_build_object('Digest', 'legacy'))) WHERE id=$1`, owner.ID, retained.ComputeState, field) + incompatible, err := db.New(f.store.pool).HasIncompatibleRuntimeComputeState(t.Context(), sandbox.SuspensionStateVersion) + if err != nil || !incompatible { + t.Fatalf("legacy %s provenance accepted: incompatible=%v err=%v", field, incompatible, err) + } } - if !strings.Contains(string(state), `"snapshot"`) { - t.Fatal("old receipt was rewritten") + // Adapter-owned data may contain any private field names. The activation + // fence must inspect only the shared envelope, never this opaque string. + f.sql(`UPDATE runtime_allocations SET compute_state=jsonb_set($2::jsonb, '{retained,Data}', to_jsonb($3::text)) WHERE id=$1`, owner.ID, retained.ComputeState, `{"snapshot":{"Digest":"private","CheckpointID":"native","CheckpointRoot":"owned"}}`) + incompatible, err := db.New(f.store.pool).HasIncompatibleRuntimeComputeState(t.Context(), sandbox.SuspensionStateVersion) + if err != nil || incompatible { + t.Fatalf("opaque native payload affected activation: incompatible=%v err=%v", incompatible, err) } f.sql(`UPDATE runtime_allocations SET compute_state=$2::jsonb WHERE id=$1`, owner.ID, retained.ComputeState) f.start() diff --git a/services/core/tests/integration/runtime_idle_clock_test.go b/services/core/tests/integration/runtime_idle_clock_test.go index d2cfdffc5..715005a24 100644 --- a/services/core/tests/integration/runtime_idle_clock_test.go +++ b/services/core/tests/integration/runtime_idle_clock_test.go @@ -29,6 +29,9 @@ func managedIdleClockFixture(t *testing.T) (*Store, *Store, deployment.Allocatio if err != nil { t.Fatal(err) } + if err := reserveSessionPlacement(t, s, w, session); err != nil { + t.Fatal(err) + } owner, err := deploymentExecution(t, w).ReserveAllocation(t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: session.Environment.ID}, d.InstallationID, runtimedevice.HashCredential("runtime")) if err != nil { t.Fatal(err) @@ -68,11 +71,11 @@ func verifyManagedIdleClock(t *testing.T, s, w *Store, owner deployment.Allocati t.Fatal("idle clock did not use committed terminal ingestion", activity, before, after, err) } until := runtimeDatabaseTime(t, s).Add(time.Hour) - if _, err := deploymentExecution(t, w).SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, idleTimeout); !errors.Is(err, deployment.ErrAllocationConflict) { + if _, err := deploymentExecution(t, w).SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, idleTimeout); !errors.Is(err, deployment.ErrNotIdle) { t.Fatal("new completion admitted premature idle", err) } // Advance only the internal activity age; the remote public timestamp remains unchanged. - runtimeSuspensionSQL(t, s.pool, "UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '2 minutes' WHERE id=$1", owner.ID) + runtimeSuspensionSQL(t, s.pool, "UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '2 minutes',compute_phase_changed_at=clock_timestamp()-interval '2 minutes' WHERE id=$1", owner.ID) if _, err := deploymentExecution(t, w).SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, idleTimeout); err != nil { t.Fatal("remote timestamp delayed elapsed idle timer", err) } diff --git a/services/core/tests/integration/runtime_initialization_test.go b/services/core/tests/integration/runtime_initialization_test.go index 636474b84..b50a949c5 100644 --- a/services/core/tests/integration/runtime_initialization_test.go +++ b/services/core/tests/integration/runtime_initialization_test.go @@ -121,8 +121,22 @@ func TestEnvironmentInitializationCompletionUnknownAndRestart(t *testing.T) { if int(p.writes.Load()) != expectedSteps { t.Fatal("completed preparation replayed") } - } else if p.writes.Load() != 1 { - t.Fatal("unknown operation replayed", p.writes.Load()) + } else { + if p.writes.Load() != 1 { + t.Fatal("unknown operation replayed", p.writes.Load()) + } + stop() + _, _ = managedWorkerMode(t, s, key, p, true) + if err := p.connect(sandbox.Bootstrap{DeviceID: owner.DeviceID, Credential: p.credential}); err != nil { + t.Fatal(err) + } + time.Sleep(350 * time.Millisecond) + if p.writes.Load() != 1 || initializationState(t, s, tenant, env.ID) != "failed" { + t.Fatal("reconnect or Worker restart replayed unknown initialization") + } + if _, err := sessionAdapter(s).GetSessionExecutionBinding(t.Context(), tenant, session.ID); !errors.Is(err, sessions.ErrNotFound) { + t.Fatal("unknown initialization admitted execution", err) + } } p.mu.Lock() kills := p.kills diff --git a/services/core/tests/integration/runtime_lifecycle_nodes_test.go b/services/core/tests/integration/runtime_lifecycle_nodes_test.go index 72f26498a..b87368a27 100644 --- a/services/core/tests/integration/runtime_lifecycle_nodes_test.go +++ b/services/core/tests/integration/runtime_lifecycle_nodes_test.go @@ -28,10 +28,10 @@ func lifecycleTestNode(t *testing.T, s *Store) string { onlineManagerNode(t, s, id) return id } -func lifecycleTestSession(t *testing.T, s *Store, node string) (string, sessions.Session) { +func lifecycleTestSession(t *testing.T, s, w *Store, node string) (string, sessions.Session) { t.Helper() tenant := uuid.NewString() - session, err := createSessionOnNode(t, s, tenant, managerSessionInput(uuid.NewString()), node) + session, err := createSessionOnNode(t, s, w, tenant, managerSessionInput(uuid.NewString()), node) if err != nil { t.Fatal(err) } @@ -39,7 +39,7 @@ func lifecycleTestSession(t *testing.T, s *Store, node string) (string, sessions } func lifecycleTestAllocation(t *testing.T, s, w *Store, d managerNode, node string) deployment.Allocation { t.Helper() - tenant, session := lifecycleTestSession(t, s, node) + tenant, session := lifecycleTestSession(t, s, w, node) allocation, err := deploymentExecution(t, w).ReserveAllocation(t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: session.Environment.ID}, d.InstallationID, runtimedevice.HashCredential(uuid.NewString())) if err != nil { t.Fatal(err) @@ -53,11 +53,11 @@ func TestRuntimeLifecycleNodePagesAreIndependent(t *testing.T) { var allocated, pending []string for range 34 { allocated = append(allocated, lifecycleTestAllocation(t, s, w, d, d.NodeID).ID) - _, session := lifecycleTestSession(t, s, d.NodeID) + _, session := lifecycleTestSession(t, s, w, d.NodeID) pending = append(pending, session.Environment.ID) } second := lifecycleTestAllocation(t, s, w, d, other) - _, secondPending := lifecycleTestSession(t, s, other) + _, secondPending := lifecycleTestSession(t, s, w, other) // Offline and unresolved cleanup remain discoverable without changing placement. if _, err := s.pool.Exec(t.Context(), "UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1", d.NodeID); err != nil { t.Fatal(err) @@ -126,7 +126,7 @@ func TestRuntimeLifecycleNodePagesAreIndependent(t *testing.T) { func TestRuntimeLifecycleNodeInventoryAndRouting(t *testing.T) { s, w, d := managerFixture(t, 100, 100) other := lifecycleTestNode(t, s) - tenant, session := lifecycleTestSession(t, s, other) + tenant, session := lifecycleTestSession(t, s, w, other) environment := session.Environment.ID checkRoute := func(want string, wantErr error) { t.Helper() @@ -173,7 +173,7 @@ func TestRuntimeLifecycleNodeInventoryAndRouting(t *testing.T) { if _, err = deploymentExecution(t, w).ReleaseAllocation(t.Context(), owner); err != nil { t.Fatal(err) } - checkRoute(other, nil) + checkRoute("", placement.ErrNodeUnavailable) // Released ownership is no longer a lifecycle route. if rows, err := deploymentStore(w).LifecycleAllocations(t.Context(), other, ""); err != nil || len(rows) != 0 { t.Fatal("released allocation scanned", rows, err) } @@ -190,7 +190,7 @@ func TestRuntimeLifecycleNodeRejectsMissingOrReleasedPlacement(t *testing.T) { for _, mutation := range []string{"DELETE FROM runtime_placements WHERE environment_id=$1", "UPDATE runtime_placements SET released_at=clock_timestamp() WHERE environment_id=$1"} { t.Run(mutation[:6], func(t *testing.T) { s, w, d := managerFixture(t, 4, 4) - tenant, session := lifecycleTestSession(t, s, d.NodeID) + tenant, session := lifecycleTestSession(t, s, w, d.NodeID) if _, err := s.pool.Exec(t.Context(), mutation, session.Environment.ID); err != nil { t.Fatal(err) } diff --git a/services/core/tests/integration/runtime_node_generations_test.go b/services/core/tests/integration/runtime_node_generations_test.go index f491d98cf..3197a00c7 100644 --- a/services/core/tests/integration/runtime_node_generations_test.go +++ b/services/core/tests/integration/runtime_node_generations_test.go @@ -29,14 +29,21 @@ func changeNodeTarget(t *testing.T, w *Store, view deployment.View, input sandbo func generationHeartbeat(t *testing.T, s *Store, node deployment.Enrollment, connection string, view deployment.View, state string) { t.Helper() - err := deploymentService(t, s).HeartbeatGenerations(t.Context(), node.NodeID, connection, view.OwnerEpoch, deployment.NodeHealth{}, []sandbox.GenerationStatus{{Generation: view.Generation, SpecificationDigest: view.SpecificationDigest, State: state}}) + status := sandbox.GenerationStatus{Generation: view.Generation, SpecificationDigest: view.SpecificationDigest, State: state} + if state == "ready" && view.Provider == "microsandbox" { + status.Checkpoint = &sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"} + } + err := deploymentService(t, s).HeartbeatGenerations(t.Context(), node.NodeID, connection, view.OwnerEpoch, deployment.NodeHealth{}, []sandbox.GenerationStatus{status}) if err != nil { t.Fatal(err) } } -func placedGeneration(t *testing.T, s *Store, session sessions.Session) (string, int64) { +func placedGeneration(t *testing.T, s, w *Store, session sessions.Session) (string, int64) { t.Helper() + if err := reserveSessionPlacement(t, s, w, session); err != nil { + t.Fatal(err) + } var node string var generation int64 if err := s.pool.QueryRow(t.Context(), "SELECT node_id::text,deployment_generation FROM runtime_placements WHERE environment_id=$1", session.Environment.ID).Scan(&node, &generation); err != nil { @@ -59,7 +66,7 @@ func TestNodeGenerationsCapacityFallbackAndImmutablePending(t *testing.T) { if err != nil { t.Fatal(err) } - pendingNode, pendingGeneration := placedGeneration(t, s, pending) + pendingNode, pendingGeneration := placedGeneration(t, s, w, pending) token, err := EnrollmentTestToken(nodes.CreateEnrollment(t.Context(), deployment.Capacity{MaxActive: 1, MaxRetained: 2})) if err != nil { t.Fatal(err) @@ -76,7 +83,7 @@ func TestNodeGenerationsCapacityFallbackAndImmutablePending(t *testing.T) { if err != nil { t.Fatal("target preparation suppressed fallback", err) } - if _, g := placedGeneration(t, s, fallback); g != 1 { + if _, g := placedGeneration(t, s, w, fallback); g != 1 { t.Fatal(g) } generationHeartbeat(t, s, a, ca, second, "ready") @@ -84,7 +91,7 @@ func TestNodeGenerationsCapacityFallbackAndImmutablePending(t *testing.T) { if err != nil { t.Fatal(err) } - if n, g := placedGeneration(t, s, newest); n != a.NodeID || g != 2 { + if n, g := placedGeneration(t, s, w, newest); n != a.NodeID || g != 2 { t.Fatal(n, g) } // Capacity remains shared across generations. A full newest pin cannot hide B. @@ -95,7 +102,7 @@ func TestNodeGenerationsCapacityFallbackAndImmutablePending(t *testing.T) { if err != nil { t.Fatal(err) } - if n, g := placedGeneration(t, s, old); n != b.NodeID || g != 1 { + if n, g := placedGeneration(t, s, w, old); n != b.NodeID || g != 1 { t.Fatal("newest full hid older free node", n, g) } third, _ := changeNodeTarget(t, w, second, input) @@ -163,10 +170,14 @@ func TestNodeGenerationsReconnectAndV1Fallback(t *testing.T) { if err != nil || nodes[0].ProviderReady || *nodes[0].Rollout.ReadyGeneration != 1 { t.Fatal("reconnect inherited readiness or lost pin", nodes, err) } - if _, err = s.CreateSession(t.Context(), uuid.NewString(), managerSessionInput(uuid.NewString())); !errors.Is(err, placement.ErrNodeUnavailable) { + queued, err := s.CreateSession(t.Context(), uuid.NewString(), managerSessionInput(uuid.NewString())) + if err != nil { + t.Fatal(err) + } + if err := reserveSessionPlacement(t, s, w, queued); !errors.Is(err, placement.ErrNodeUnavailable) { t.Fatal("unconfirmed connection admitted", err) } - if err = service.Heartbeat(t.Context(), node.NodeID, connection, first.OwnerEpoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err = service.Heartbeat(t.Context(), node.NodeID, connection, first.OwnerEpoch, deployment.NodeHealth{ProviderReady: true}, nil); err != nil { t.Fatal(err) } if _, err = s.CreateSession(t.Context(), uuid.NewString(), managerSessionInput(uuid.NewString())); err != nil { @@ -175,7 +186,7 @@ func TestNodeGenerationsReconnectAndV1Fallback(t *testing.T) { } func TestNodeGenerationPreparationRefusalCreatesNoProvisionalOwnership(t *testing.T) { - s, _, first, _ := webSpecificationFixture(t, "docker") + s, w, first, _ := webSpecificationFixture(t, "docker") node := specificationNode(t, s, first) connection := uuid.NewString() if err := deploymentService(t, s).ConnectNode(t.Context(), node.NodeID, connection, first.OwnerEpoch); err != nil { @@ -183,15 +194,23 @@ func TestNodeGenerationPreparationRefusalCreatesNoProvisionalOwnership(t *testin } generationHeartbeat(t, s, node, connection, first, "preparing") tenant := uuid.NewString() - if _, err := s.CreateSession(t.Context(), tenant, managerSessionInput(uuid.NewString())); !errors.Is(err, placement.ErrNodesPreparing) { + queued, err := s.CreateSession(t.Context(), tenant, managerSessionInput(uuid.NewString())) + if err != nil { + t.Fatal(err) + } + if err := reserveSessionPlacement(t, s, w, queued); !errors.Is(err, placement.ErrNodesPreparing) { t.Fatal("actual preparation was not identified", err) } var sessions, placements int - if err := s.pool.QueryRow(t.Context(), "SELECT (SELECT count(*) FROM sessions WHERE tenant_id=$1),(SELECT count(*) FROM runtime_placements)", tenant).Scan(&sessions, &placements); err != nil || sessions != 0 || placements != 0 { - t.Fatal("refusal left provisional ownership", sessions, placements, err) + if err := s.pool.QueryRow(t.Context(), "SELECT (SELECT count(*) FROM sessions WHERE tenant_id=$1),(SELECT count(*) FROM runtime_placements)", tenant).Scan(&sessions, &placements); err != nil || sessions != 1 || placements != 0 { + t.Fatal("waiting request consumed provisional compute", sessions, placements, err) } generationHeartbeat(t, s, node, connection, first, "failed") - if _, err := s.CreateSession(t.Context(), tenant, managerSessionInput(uuid.NewString())); !errors.Is(err, placement.ErrNodeUnavailable) { + if err := reserveSessionPlacement(t, s, w, queued); !errors.Is(err, placement.ErrNodeUnavailable) { t.Fatal("failed preparation advertised active work", err) } + generationHeartbeat(t, s, node, connection, first, "ready") + if err := reserveSessionPlacement(t, s, w, queued); err != nil { + t.Fatal("ready node did not admit waiting Session", err) + } } diff --git a/services/core/tests/integration/runtime_node_lifecycle_fixture_test.go b/services/core/tests/integration/runtime_node_lifecycle_fixture_test.go index 837cb4279..c9d0f2e07 100644 --- a/services/core/tests/integration/runtime_node_lifecycle_fixture_test.go +++ b/services/core/tests/integration/runtime_node_lifecycle_fixture_test.go @@ -77,6 +77,7 @@ type nodeIsolationFixture struct { nodes *deployment.Service pool *pgxpool.Pool worker *execution.Worker + placement *deployment.ExecutionOperations provider *nodeIsolationProvider key, nodeA, nodeB string epoch uint64 @@ -119,7 +120,9 @@ func newNodeIsolationFixture(t *testing.T, mode string) *nodeIsolationFixture { f := &nodeIsolationFixture{initializationCancel: cancelPreparation, t: t, store: s, nodes: deploymentService(t, s), pool: pool, provider: p, key: webDeployment(t, s, "microsandbox"), nodeA: uuid.NewString(), nodeB: uuid.NewString()} // Keep restored compute awake throughout the isolation assertions. // The suspension setup explicitly dates its activity two minutes in the past. - f.worker = startWebWorker(t, s, registry, f.key, p, &execution.RuntimeSuspensionPolicy{IdleTimeout: time.Minute, Retention: time.Hour}) + owner := executionOwner(t, s) + f.placement = owner.Deployment + f.worker = startOwnedWorker(t, t.Context(), s, &execution.Dispatcher{Registry: registry, ManagedRuntimes: webRuntimes(t, s, f.key, p, &execution.RuntimeSuspensionPolicy{IdleTimeout: time.Minute, Retention: time.Hour})}, owner) t.Cleanup(f.stop) f.epoch = fixtureOwnerEpoch(t, s) f.enroll(f.nodeA) @@ -145,7 +148,7 @@ func (f *nodeIsolationFixture) online(id string) { if err := f.nodes.ConnectNode(f.t.Context(), id, connection, f.epoch); err != nil { f.t.Fatal(err) } - if err := f.nodes.Heartbeat(f.t.Context(), id, connection, f.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := fixtureNodeHeartbeat(f.t.Context(), f.store, f.nodes, id, connection, f.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { f.t.Fatal(err) } } @@ -180,13 +183,16 @@ func (f *nodeIsolationFixture) session(node string, initialize bool) (string, se f.t.Fatal(err) } for _, value := range others { - if err := f.nodes.Heartbeat(f.t.Context(), value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: false}); err != nil { + if err := fixtureNodeHeartbeat(f.t.Context(), f.store, f.nodes, value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: false}); err != nil { f.t.Fatal(err) } } session, err := f.store.CreateSession(f.t.Context(), tenant, input) + if err == nil { + _, err = f.placement.EnsurePlacement(f.t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: session.Environment.ID}, f.key) + } for _, value := range others { - if err := f.nodes.Heartbeat(context.WithoutCancel(f.t.Context()), value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := fixtureNodeHeartbeat(context.WithoutCancel(f.t.Context()), f.store, f.nodes, value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { f.t.Fatal(err) } } diff --git a/services/core/tests/integration/runtime_node_lifecycle_test.go b/services/core/tests/integration/runtime_node_lifecycle_test.go index 545e87866..e601d41b4 100644 --- a/services/core/tests/integration/runtime_node_lifecycle_test.go +++ b/services/core/tests/integration/runtime_node_lifecycle_test.go @@ -24,7 +24,7 @@ func TestManagedNodesIsolateBlockedProviderAndInitialization(t *testing.T) { if _, err := f.pool.Exec(t.Context(), "INSERT INTO turns(id,session_id,status,completed_at) VALUES($1,$2,'completed',clock_timestamp()-interval '2 minutes')", uuid.NewString(), wakeOwner.SessionID); err != nil { t.Fatal(err) } - if _, err := f.pool.Exec(t.Context(), "UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '2 minutes' WHERE id=$1", wakeOwner.ID); err != nil { + if _, err := f.pool.Exec(t.Context(), "UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '2 minutes',compute_phase_changed_at=clock_timestamp()-interval '2 minutes' WHERE id=$1", wakeOwner.ID); err != nil { t.Fatal(err) } f.phase(wakeTenant, wakeEnv.ID, "suspended") diff --git a/services/core/tests/integration/runtime_nodes_test.go b/services/core/tests/integration/runtime_nodes_test.go index 3ab2694c2..2d83b955e 100644 --- a/services/core/tests/integration/runtime_nodes_test.go +++ b/services/core/tests/integration/runtime_nodes_test.go @@ -7,11 +7,11 @@ import ( "strings" "sync" "testing" - "time" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment/placement" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/runtimedevice" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" "github.com/google/uuid" "github.com/jackc/pgx/v5" @@ -34,7 +34,7 @@ func onlineManagerNode(t *testing.T, s *Store, id string) string { if err := nodes.ConnectNode(t.Context(), id, connection, managerEpoch(t, s)); err != nil { t.Fatal(err) } - if err := nodes.Heartbeat(t.Context(), id, connection, managerEpoch(t, s), deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := fixtureNodeHeartbeat(t.Context(), s, nodes, id, connection, managerEpoch(t, s), deployment.NodeHealth{ProviderReady: true}); err != nil { t.Fatal(err) } return connection @@ -44,8 +44,8 @@ func managerSessionInput(key string) sessions.CreateSession { } // createSessionOnNode steers automatic placement in multi-node tests: only node -// stays provider-ready while the Session is created. -func createSessionOnNode(t *testing.T, s *Store, tenant string, input sessions.CreateSession, node string) (sessions.Session, error) { +// stays provider-ready while the scheduler reserves the created Session. +func createSessionOnNode(t *testing.T, s, w *Store, tenant string, input sessions.CreateSession, node string) (sessions.Session, error) { t.Helper() // Use the same authenticated readiness observations as the scheduler. The // compatibility provider_ready column alone is not admission authority. @@ -72,18 +72,32 @@ func createSessionOnNode(t *testing.T, s *Store, tenant string, input sessions.C } nodes := deploymentService(t, s) for _, value := range others { - if err := nodes.Heartbeat(t.Context(), value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: false}); err != nil { + if err := fixtureNodeHeartbeat(t.Context(), s, nodes, value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: false}); err != nil { t.Fatal(err) } } defer func() { for _, value := range others { - if err := nodes.Heartbeat(context.WithoutCancel(t.Context()), value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := fixtureNodeHeartbeat(context.WithoutCancel(t.Context()), s, nodes, value.id, value.connection, value.epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { t.Fatal(err) } } }() - return s.CreateSession(t.Context(), tenant, input) + session, err := s.CreateSession(t.Context(), tenant, input) + if err == nil { + err = reserveSessionPlacement(t, s, w, session) + } + return session, err +} + +func reserveSessionPlacement(t *testing.T, s, w *Store, session sessions.Session) error { + t.Helper() + setup, err := deploymentService(t, s).Setup(t.Context()) + if err != nil { + return err + } + _, err = deploymentExecution(t, w).EnsurePlacement(t.Context(), deployment.AllocationKey{TenantID: session.TenantID, EnvironmentID: session.Environment.ID}, setup.InstallationID) + return err } type sessionPlacement struct { @@ -115,7 +129,7 @@ func sessionRuntimePlacement(ctx context.Context, s *Store, tenant, session stri return placement, err } func TestRuntimeNodesAtomicPlacementAndRetry(t *testing.T) { - s, _, d := managerFixture(t, 1, 4) + s, w, d := managerFixture(t, 1, 4) tenant := uuid.NewString() var wg sync.WaitGroup results := make(chan sessions.Session, 16) @@ -132,23 +146,46 @@ func TestRuntimeNodesAtomicPlacementAndRetry(t *testing.T) { wg.Wait() close(results) close(failures) - successes := 0 for err := range failures { - if err == nil { - successes++ - } else if !errors.Is(err, placement.ErrNodeUnavailable) { - t.Fatal(err) + if err != nil { + t.Fatal("capacity rejected durable Session creation", err) } } - if successes != 1 { - t.Fatal("overbooked node", successes) + var created []sessions.Session + for session := range results { + created = append(created, session) + } + if len(created) != 16 { + t.Fatal("lost accepted Sessions", len(created)) + } + type admission struct { + session sessions.Session + err error + } + reserved := make(chan admission, len(created)) + for _, session := range created { + wg.Add(1) + go func() { defer wg.Done(); reserved <- admission{session, reserveSessionPlacement(t, s, w, session)} }() } + wg.Wait() + close(reserved) var retained sessions.Session - for session := range results { - if session.ID != "" { - retained = session + var waiting []sessions.Session + for result := range reserved { + if result.err == nil { + if retained.ID != "" { + t.Fatal("overbooked node") + } + retained = result.session + } else if errors.Is(result.err, placement.ErrNodeUnavailable) { + waiting = append(waiting, result.session) + } else { + t.Fatal(result.err) } } + if retained.ID == "" || len(waiting) != 15 { + t.Fatal("capacity admission", retained.ID, len(waiting)) + } service := deploymentService(t, s) nodes, err := service.ListNodes(t.Context()) if err != nil || len(nodes) != 1 || nodes[0].Active != 1 || nodes[0].Retained != 1 || nodes[0].Reserved != 1 { @@ -160,11 +197,20 @@ func TestRuntimeNodesAtomicPlacementAndRetry(t *testing.T) { if err := sessionService(t, s).DeleteSession(t.Context(), sessions.DeleteSessionCommand{TenantID: tenant, SessionID: retained.ID}); err != nil { t.Fatal(err) } + if err := reserveSessionPlacement(t, s, w, waiting[0]); err != nil { + t.Fatal("released slot did not admit waiting Session", err) + } + if err := sessionService(t, s).DeleteSession(t.Context(), sessions.DeleteSessionCommand{TenantID: tenant, SessionID: waiting[0].ID}); err != nil { + t.Fatal(err) + } input := managerSessionInput("retry") first, err := s.CreateSession(t.Context(), tenant, input) if err != nil { t.Fatal(err) } + if err := reserveSessionPlacement(t, s, w, first); err != nil { + t.Fatal(err) + } if _, err := s.pool.Exec(t.Context(), "UPDATE runtime_nodes SET connection_id=NULL WHERE id=$1", d.NodeID); err != nil { t.Fatal(err) } @@ -222,10 +268,14 @@ func TestRuntimeNodesEnrollmentAndEpoch(t *testing.T) { if next := managerEpoch(t, s); next != epoch+1 { t.Fatal(next) } - if err := nodes.Heartbeat(t.Context(), input.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); !errors.Is(err, deployment.ErrNodeCredential) { + if err := fixtureNodeHeartbeat(t.Context(), s, nodes, input.NodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); !errors.Is(err, deployment.ErrNodeCredential) { t.Fatal("old epoch heartbeat revived node", err) } - if _, err := s.CreateSession(t.Context(), uuid.NewString(), managerSessionInput("stale")); !errors.Is(err, placement.ErrNodeUnavailable) { + queued, err := s.CreateSession(t.Context(), uuid.NewString(), managerSessionInput("stale")) + if err != nil { + t.Fatal(err) + } + if err := reserveSessionPlacement(t, s, w, queued); !errors.Is(err, placement.ErrNodeUnavailable) { t.Fatal("stale node admitted", err) } onlineManagerNode(t, s, d.NodeID) @@ -244,6 +294,9 @@ func TestRuntimeNodesRetention(t *testing.T) { if err != nil { t.Fatal(err) } + if err := reserveSessionPlacement(t, s, w, first); err != nil { + t.Fatal(err) + } retained, err := deploymentExecution(t, w).ReserveAllocation(t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: first.Environment.ID}, next.InstallationID, runtimedevice.HashCredential("runtime")) if err != nil { t.Fatal(err) @@ -280,12 +333,17 @@ func TestRuntimeNodesRetention(t *testing.T) { } } func TestRuntimeNodesRestoreAndCreationShareCapacity(t *testing.T) { - s, w, d := managerFixture(t, 1, 4) + s, w, view, _ := webSpecificationFixture(t, "microsandbox") + node := enrollNode(t, s, view, deployment.Capacity{MaxActive: 1, MaxRetained: 4}) + d := managerNode{InstallationID: view.InstallationID, NodeID: node.NodeID} tenant := uuid.NewString() session, err := s.CreateSession(t.Context(), tenant, managerSessionInput("first")) if err != nil { t.Fatal(err) } + if err := reserveSessionPlacement(t, s, w, session); err != nil { + t.Fatal(err) + } allocation, err := deploymentExecution(t, w).ReserveAllocation(t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: session.Environment.ID}, d.InstallationID, runtimedevice.HashCredential("runtime")) if err != nil { t.Fatal(err) @@ -297,18 +355,23 @@ func TestRuntimeNodesRestoreAndCreationShareCapacity(t *testing.T) { if err != nil { t.Fatal(err) } - until := time.Now().Add(time.Hour) + if err := deploymentService(t, s).TouchActivity(t.Context(), allocation.TenantID, allocation.EnvironmentID); err != nil { + t.Fatal(err) + } + second, err := s.CreateSession(t.Context(), tenant, managerSessionInput("second")) + if err != nil { + t.Fatal(err) + } start := make(chan struct{}) results := make(chan error, 2) go func() { <-start - _, err := deploymentExecution(t, w).SetCompute(t.Context(), allocation, "restoring", json.RawMessage(`{"target":{"id":"restore"}}`), &until, 0) + _, err := deploymentExecution(t, w).BeginRestore(t.Context(), allocation, sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, json.RawMessage(`{"target":{"id":"restore"}}`)) results <- err }() go func() { <-start - _, err := s.CreateSession(t.Context(), tenant, managerSessionInput("second")) - results <- err + results <- reserveSessionPlacement(t, s, w, second) }() close(start) success := 0 @@ -344,6 +407,9 @@ func TestRuntimeNodesLongOfflineRetainsExactAllocation(t *testing.T) { if err != nil { t.Fatal(err) } + if err := reserveSessionPlacement(t, s, w, session); err != nil { + t.Fatal(err) + } owner, err := deploymentExecution(t, w).ReserveAllocation(t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: session.Environment.ID}, d.InstallationID, runtimedevice.HashCredential("runtime")) if err != nil { t.Fatal(err) diff --git a/services/core/tests/integration/runtime_observation_test.go b/services/core/tests/integration/runtime_observation_test.go index 5ad2acf2b..1c84c5c95 100644 --- a/services/core/tests/integration/runtime_observation_test.go +++ b/services/core/tests/integration/runtime_observation_test.go @@ -16,6 +16,9 @@ func TestRuntimeNodeObservationRetainsResourcesAndFencesStaleResults(t *testing. if err != nil { t.Fatal(err) } + if err := reserveSessionPlacement(t, s, w, session); err != nil { + t.Fatal(err) + } owner, err := deploymentExecution(t, w).ReserveAllocation(t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: session.Environment.ID}, d.InstallationID, runtimedevice.HashCredential("runtime")) if err != nil { t.Fatal(err) diff --git a/services/core/tests/integration/runtime_suspension_test.go b/services/core/tests/integration/runtime_suspension_test.go index 3e2a9cf4c..a85b5e072 100644 --- a/services/core/tests/integration/runtime_suspension_test.go +++ b/services/core/tests/integration/runtime_suspension_test.go @@ -10,6 +10,7 @@ import ( "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/runtimedevice" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sessions" "github.com/google/uuid" "github.com/jackc/pgx/v5/pgxpool" @@ -57,7 +58,13 @@ func runtimeSuspensionStep(t *testing.T, w *Store, owner deployment.Allocation, if owner.ComputePhase == "running" && phase == "quiescing" { idleTimeout = time.Nanosecond } - next, err := deploymentExecution(t, w).SetCompute(t.Context(), owner, phase, json.RawMessage(`{"instance":"original","snapshot":"qualified"}`), until, idleTimeout) + var next deployment.Allocation + var err error + if phase == "restoring" { + next, err = deploymentExecution(t, w).BeginRestore(t.Context(), owner, sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, json.RawMessage(`{"instance":"original","snapshot":"qualified"}`)) + } else { + next, err = deploymentExecution(t, w).SetCompute(t.Context(), owner, phase, json.RawMessage(`{"instance":"original","snapshot":"qualified"}`), until, idleTimeout) + } if err != nil { t.Fatalf("%s -> %s: %v", owner.ComputePhase, phase, err) } @@ -65,12 +72,12 @@ func runtimeSuspensionStep(t *testing.T, w *Store, owner deployment.Allocation, } func TestRuntimeSuspensionRequiresIdleAndNoPendingWork(t *testing.T) { - cases := []string{"no_completed_turn", "queued", "in_progress", "waiting", "subagent_queued", "subagent_in_progress", "subagent_waiting", "input_reservation", "file_write", "idle"} + cases := []string{"no_completed_turn", "initial_input", "queued", "in_progress", "waiting", "subagent_queued", "subagent_in_progress", "subagent_waiting", "input_reservation", "file_write", "idle"} for _, kind := range cases { t.Run(kind, func(t *testing.T) { _, w, pool, owner := runtimeSuspensionFixture(t) completed := "" - if kind != "no_completed_turn" { + if kind != "no_completed_turn" && kind != "initial_input" { completed = runtimeSuspensionCompleted(t, pool, owner) } switch kind { @@ -81,6 +88,8 @@ func TestRuntimeSuspensionRequiresIdleAndNoPendingWork(t *testing.T) { runtimeSuspensionSQL(t, pool, `INSERT INTO turn_events(session_id,turn_id,ordinal,kind,payload) VALUES($1,$2,1,'subagent','{}')`, owner.SessionID, completed) runtimeSuspensionSQL(t, pool, `INSERT INTO subagent_identities(id,session_id,device_id,engine,native_id,parent_native_id,native_created_at,first_turn_id,first_event_ordinal) VALUES($1,$2,$3,'codex','child','root',1,$4,1)`, child, owner.SessionID, owner.DeviceID, completed) runtimeSuspensionSQL(t, pool, `INSERT INTO subagent_turns(id,session_id,subagent_id,native_id,status,created_at) VALUES($1,$2,$3,'child-turn',$4,clock_timestamp())`, uuid.NewString(), owner.SessionID, child, strings.TrimPrefix(kind, "subagent_")) + case "initial_input": + runtimeSuspensionSQL(t, pool, `INSERT INTO environment_input_reservations(id,session_id,idempotency_key,batch,is_initial,created_at,deadline) VALUES($1,$2,'initial','[{}]',true,clock_timestamp(),clock_timestamp()+interval '1 minute')`, uuid.NewString(), owner.SessionID) case "input_reservation": runtimeSuspensionSQL(t, pool, `INSERT INTO environment_input_reservations(id,session_id,idempotency_key,batch,created_at,deadline) VALUES($1,$2,'pending','[{}]',clock_timestamp(),clock_timestamp()+interval '1 minute')`, uuid.NewString(), owner.SessionID) case "file_write": @@ -243,7 +252,7 @@ func TestRuntimeSuspensionRetentionAndDeletedSession(t *testing.T) { t.Fatal("snapshot retention expiry not observed", expired, err) } // Use the earlier unexpired observation to exercise expiry at the database CAS. - if _, err := deploymentExecution(t, w).SetCompute(t.Context(), retained, "restoring", json.RawMessage(`{}`), &until, 0); !errors.Is(err, deployment.ErrAllocationConflict) { + if _, err := deploymentExecution(t, w).BeginRestore(t.Context(), retained, sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, json.RawMessage(`{}`)); !errors.Is(err, deployment.ErrAllocationConflict) { t.Fatal("expired snapshot restored from stale observation", err) } if err := sessionService(t, s).DeleteSession(t.Context(), sessions.DeleteSessionCommand{TenantID: owner.TenantID, SessionID: owner.SessionID}); err != nil { @@ -256,7 +265,7 @@ func TestRuntimeSuspensionRetentionAndDeletedSession(t *testing.T) { if err != nil || !deleted.SessionDeleted || deleted.ComputeWakeRequested { t.Fatal("deleted session was woken", deleted, err) } - if _, err := deploymentExecution(t, w).SetCompute(t.Context(), deleted, "restoring", json.RawMessage(`{}`), &until, 0); !errors.Is(err, sessions.ErrNotFound) { + if _, err := deploymentExecution(t, w).BeginRestore(t.Context(), deleted, sandbox.CheckpointCompatibility{ArtifactDomain: "fixture-store", ExecutionClass: "fixture-runtime"}, json.RawMessage(`{}`)); !errors.Is(err, sessions.ErrNotFound) { t.Fatal("deleted session restored", err) } } @@ -305,7 +314,7 @@ func TestRuntimeSuspensionCountsUncertainCapacityUntilReleased(t *testing.T) { func TestRuntimeSuspensionIdleStartsAfterLastCompletion(t *testing.T) { s, w, pool, owner := runtimeSuspensionFixture(t) runtimeSuspensionCompleted(t, pool, owner) - runtimeSuspensionSQL(t, pool, `UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '2 hours' WHERE id=$1`, owner.ID) + runtimeSuspensionSQL(t, pool, `UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '2 hours',compute_phase_changed_at=clock_timestamp()-interval '2 hours' WHERE id=$1`, owner.ID) before := runtimeDatabaseTime(t, s) id, _ := parseID(owner.SessionID) if err := s.queries.RecordRuntimeTerminalActivity(t.Context(), id); err != nil { @@ -371,7 +380,7 @@ func TestRuntimeSuspensionRechecksCompletionAgainstIdleTimeout(t *testing.T) { t.Run(kind, func(t *testing.T) { s, w, pool, owner := runtimeSuspensionFixture(t) turn := runtimeSuspensionCompleted(t, pool, owner) - runtimeSuspensionSQL(t, pool, `UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '20 minutes' WHERE id=$1`, owner.ID) + runtimeSuspensionSQL(t, pool, `UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '20 minutes',compute_phase_changed_at=clock_timestamp()-interval '20 minutes' WHERE id=$1`, owner.ID) var err error owner, err = deploymentStore(s).EnvironmentAllocation(t.Context(), deployment.AllocationKey{TenantID: owner.TenantID, EnvironmentID: owner.EnvironmentID}) if err != nil { @@ -402,7 +411,7 @@ func TestRuntimeSuspensionRechecksCompletionAgainstIdleTimeout(t *testing.T) { } } until := time.Now().Add(time.Hour) - if _, err := deploymentExecution(t, w).SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, idleTimeout); !errors.Is(err, deployment.ErrAllocationConflict) { + if _, err := deploymentExecution(t, w).SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, idleTimeout); !errors.Is(err, deployment.ErrNotIdle) && !errors.Is(err, deployment.ErrAllocationConflict) { t.Fatal("completion after idle observation did not fence quiesce", err) } activity, err := deploymentStore(w).Activity(t.Context(), owner.ID) @@ -459,6 +468,9 @@ func TestRuntimeComputePhaseChangedAtInNodeAllocations(t *testing.T) { if err != nil { t.Fatal(err) } + if err := reserveSessionPlacement(t, s, w, session); err != nil { + t.Fatal(err) + } allocation, err := deploymentExecution(t, w).ReserveAllocation(t.Context(), deployment.AllocationKey{TenantID: tenant, EnvironmentID: session.Environment.ID}, d.InstallationID, runtimedevice.HashCredential("runtime")) if err != nil { t.Fatal(err) @@ -479,3 +491,32 @@ func TestRuntimeComputePhaseChangedAtInNodeAllocations(t *testing.T) { t.Fatal("an unknown phase time was not null", string(encoded)) } } + +func TestNeverStartedSessionIdleStartsAfterInitialization(t *testing.T) { + _, w, pool, owner := runtimeSuspensionFixture(t) + const timeout = 5 * time.Minute + until := time.Now().Add(time.Hour) + runtimeSuspensionSQL(t, pool, `UPDATE runtime_allocations SET compute_activity_at=clock_timestamp()-interval '1 hour',created_at=clock_timestamp()-interval '1 hour' WHERE id=$1`, owner.ID) + activity, err := deploymentStore(w).Activity(t.Context(), owner.ID) + if err != nil || activity.ReadyToSuspend(timeout) { + t.Fatal("creation age bypassed ready-time idle clock", activity, err) + } + runtimeSuspensionSQL(t, pool, `UPDATE runtime_allocations SET compute_phase_changed_at=clock_timestamp()-interval '6 minutes' WHERE id=$1`, owner.ID) + activity, err = deploymentStore(w).Activity(t.Context(), owner.ID) + if err != nil || !activity.ReadyToSuspend(timeout) { + t.Fatal("never-started initialized Session cannot idle", activity, err) + } + runtimeSuspensionSQL(t, pool, `UPDATE environments SET initialization='running' WHERE id=$1`, owner.EnvironmentID) + if _, err = deploymentExecution(t, w).SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, timeout); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("unfinished initialization admitted suspension", err) + } + runtimeSuspensionSQL(t, pool, `UPDATE environments SET initialization='complete' WHERE id=$1`, owner.EnvironmentID) + runtimeSuspensionSQL(t, pool, `UPDATE runtime_allocations SET compute_wake_requested=true WHERE id=$1`, owner.ID) + if _, err = deploymentExecution(t, w).SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, timeout); !errors.Is(err, deployment.ErrAllocationConflict) { + t.Fatal("wake lost to suspension", err) + } + runtimeSuspensionSQL(t, pool, `UPDATE runtime_allocations SET compute_wake_requested=false WHERE id=$1`, owner.ID) + if _, err = deploymentExecution(t, w).SetCompute(t.Context(), owner, "quiescing", json.RawMessage(`{}`), &until, timeout); err != nil { + t.Fatal("idle Session with no Turn was blocked", err) + } +} diff --git a/services/core/tests/integration/sandbox_deployment_audit_http_test.go b/services/core/tests/integration/sandbox_deployment_audit_http_test.go new file mode 100644 index 000000000..1f2ee5ba7 --- /dev/null +++ b/services/core/tests/integration/sandbox_deployment_audit_http_test.go @@ -0,0 +1,83 @@ +package integration + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/api" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/deployment" + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" +) + +// The HTTP boundary must supply provenance; this bridge passes its untouched +// context into the actual deployment transaction, including its audit write. +type auditedDeploymentChanges struct { + strictStandIn + operations *deployment.ExecutionOperations + installation string +} + +func (d auditedDeploymentChanges) UpdateSandboxDeployment(ctx context.Context, input sandbox.Selection) (deployment.View, error) { + return d.operations.Update(ctx, d.installation, input) +} + +func TestSandboxDeploymentUpdateHTTPCommitsWithAdministratorAudit(t *testing.T) { + store, writer, initial, input := webSpecificationFixture(t, "docker") + changes := auditedDeploymentChanges{strictStandIn{t}, deploymentExecution(t, writer), initial.InstallationID} + handler, err := publicHandler(t, store, newTestAuthenticator(t, nil), "codex", func(d *api.Dependencies) { d.Sandboxes.DeploymentChanges = changes }) + if err != nil { + t.Fatal(err) + } + call := func(generation uint64, cpus uint32, key, actor string, status int) *httptest.ResponseRecorder { + t.Helper() + resources := input.Resources + resources.CPUs = cpus + body, err := json.Marshal(map[string]any{"expected_generation": generation, "provider": "docker", "configuration": map[string]any{}, "resources": resources, "runtime": input.Runtime}) + if err != nil { + t.Fatal(err) + } + request := httptest.NewRequest(http.MethodPut, "/core/v1/sandbox/deployment", strings.NewReader(string(body))) + request.Header.Set("Authorization", "Bearer "+key) + request.Header.Set("Content-Type", "application/json") + request.Header.Set("X-Core-Console-Actor", actor) + response := httptest.NewRecorder() + handler.ServeHTTP(response, request) + if response.Code != status { + t.Fatalf("update status=%d want=%d body=%s", response.Code, status, response.Body) + } + return response + } + call(1, 3, "invalid", "operator", http.StatusUnauthorized) + response := call(1, 3, "admin", "operator", http.StatusOK) + current, err := deploymentService(t, store).View(t.Context()) + if err != nil || current.Generation != 2 || current.Specification.Resources.CPUs != 3 { + t.Fatal("deployment did not commit", current, err) + } + var credential, actor, requestID, traceID, action, kind, resource string + var project, tenant *string + err = store.pool.QueryRow(t.Context(), `SELECT admin_credential_id,actor_label,request_id,trace_id,action,resource_type,resource_id,project_id,tenant_id FROM admin_audit_log WHERE resource_type='sandbox_deployment' AND resource_id=$1`, initial.InstallationID).Scan(&credential, &actor, &requestID, &traceID, &action, &kind, &resource, &project, &tenant) + if err != nil { + t.Fatal(err) + } + digest := sha256.Sum256([]byte("admin")) + if credential != hex.EncodeToString(digest[:])[:8] || actor != "operator" || requestID == "" || requestID != response.Header().Get("x-request-id") || traceID == "" || action != "change" || kind != "sandbox_deployment" || resource != initial.InstallationID || project != nil || tenant != nil { + t.Fatal("administrator provenance differs", credential, actor, requestID, traceID, project, tenant) + } + // Invalid audit provenance still rolls the mutation back atomically. + call(2, 4, "admin", strings.Repeat("a", 129), http.StatusBadRequest) + call(1, 4, "admin", "operator", http.StatusConflict) + current, err = deploymentService(t, store).View(t.Context()) + if err != nil || current.Generation != 2 || current.Specification.Resources.CPUs != 3 { + t.Fatal("rejected update changed deployment", current, err) + } + var count int + if err = store.pool.QueryRow(t.Context(), `SELECT count(*) FROM admin_audit_log WHERE resource_type='sandbox_deployment' AND resource_id=$1`, initial.InstallationID).Scan(&count); err != nil || count != 1 { + t.Fatal("rejected request wrote audit", count, err) + } +} diff --git a/services/core/tests/integration/sandbox_deployment_worker_test.go b/services/core/tests/integration/sandbox_deployment_worker_test.go index 7e76740a7..528a351c1 100644 --- a/services/core/tests/integration/sandbox_deployment_worker_test.go +++ b/services/core/tests/integration/sandbox_deployment_worker_test.go @@ -50,8 +50,12 @@ func TestSandboxDeploymentWorkerActivatesWithoutRestart(t *testing.T) { if _, err := w.InitializeSandboxDeployment(t.Context(), sandbox.Selection{DeploymentSpec: SandboxDeploymentTestSpec("docker"), Provider: "docker"}); err != nil { t.Fatal(err) } - if _, err := s.CreateSession(t.Context(), uuid.NewString(), input); !errors.Is(err, placement.ErrNodeUnavailable) { - t.Fatal("zero-node deployment admitted Session", err) + queued, err := s.CreateSession(t.Context(), uuid.NewString(), input) + if err != nil { + t.Fatal("zero-node deployment rejected waiting Session", err) + } + if _, err := w.ProvisionEnvironment(t.Context(), queued.TenantID, queued.Environment.ID, id); !errors.Is(err, placement.ErrNodeUnavailable) { + t.Fatal("zero-node deployment provisioned compute", err) } token, err := EnrollmentTestToken(deployments.CreateEnrollment(t.Context(), deployment.Capacity{MaxActive: 4, MaxRetained: 16})) if err != nil { @@ -68,7 +72,7 @@ func TestSandboxDeploymentWorkerActivatesWithoutRestart(t *testing.T) { if err := deployments.ConnectNode(t.Context(), nodeID, connection, epoch); err != nil { t.Fatal(err) } - if err := deployments.Heartbeat(t.Context(), nodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}); err != nil { + if err := deployments.Heartbeat(t.Context(), nodeID, connection, epoch, deployment.NodeHealth{ProviderReady: true}, nil); err != nil { t.Fatal(err) } } diff --git a/services/core/tests/integration/sandbox_generations_test.go b/services/core/tests/integration/sandbox_generations_test.go index da94f3457..9932151e8 100644 --- a/services/core/tests/integration/sandbox_generations_test.go +++ b/services/core/tests/integration/sandbox_generations_test.go @@ -111,6 +111,9 @@ func TestPendingPlacementGenerationSurvivesRepeatedUpdates(t *testing.T) { s, w, view, _ := webSpecificationFixture(t, "docker") node := specificationNode(t, s, view) tenant, session := managedArchiveSession(t, s, managerSessionInput(uuid.NewString())) + if err := reserveSessionPlacement(t, s, w, session); err != nil { + t.Fatal(err) + } for generation := uint64(1); generation <= 2; generation++ { tx, err := s.pool.Begin(t.Context()) if err != nil { diff --git a/services/core/tests/integration/sandbox_reset_test.go b/services/core/tests/integration/sandbox_reset_test.go index cc37abdc4..18936ec6b 100644 --- a/services/core/tests/integration/sandbox_reset_test.go +++ b/services/core/tests/integration/sandbox_reset_test.go @@ -253,6 +253,9 @@ func TestSandboxResetAuditFailureRollsBackPauseAndCompletion(t *testing.T) { func TestSandboxResetSnapshotCountsOfflineOwnershipOnce(t *testing.T) { s, w, d := managerFixture(t, 10, 10) _, pending := managedArchiveSession(t, s, managerSessionInput(uuid.NewString())) + if err := reserveSessionPlacement(t, s, w, pending); err != nil { + t.Fatal(err) + } tenant, suspended := managedArchiveSession(t, s, managerSessionInput(uuid.NewString())) allocation := archiveAllocation(t, w, tenant, suspended, d.InstallationID) runtimeSuspensionSQL(t, s.pool, `UPDATE runtime_allocations SET compute_phase='suspended',compute_retained_until=clock_timestamp()+interval '1 hour' WHERE id=$1`, allocation.ID) diff --git a/services/core/tests/integration/sandbox_specification_lifecycle_test.go b/services/core/tests/integration/sandbox_specification_lifecycle_test.go index 6c27b7e03..f3a87dc6c 100644 --- a/services/core/tests/integration/sandbox_specification_lifecycle_test.go +++ b/services/core/tests/integration/sandbox_specification_lifecycle_test.go @@ -132,7 +132,7 @@ func TestSandboxSpecificationChangesPreserveEveryRetainedResource(t *testing.T) changes, deployments := deploymentExecution(t, w), deploymentService(t, s) node := specificationNode(t, s, view) tenant := uuid.NewString() - session, err := createSessionOnNode(t, s, tenant, managerSessionInput(uuid.NewString()), node.NodeID) + session, err := createSessionOnNode(t, s, w, tenant, managerSessionInput(uuid.NewString()), node.NodeID) if err != nil { t.Fatal(err) } @@ -310,24 +310,35 @@ func TestSandboxSpecificationAllocationRaceWithMaintenance(t *testing.T) { // A node keeps the public URL it enrolled with. After the public URL changes it // receives no new sandboxes until it is re-added. func TestNodeBoundToAnotherPublicURLGetsNoNewSandboxes(t *testing.T) { - s, _, view, _ := webSpecificationFixture(t, "docker") + s, w, view, _ := webSpecificationFixture(t, "docker") s.SetPlacement(placementRules(t, "https://old.example")) + w.SetPlacement(s.placement) node := specificationNode(t, s, view) nodes, err := deploymentService(t, s).ListNodes(t.Context()) if err != nil || len(nodes) != 1 || nodes[0].ID != node.NodeID || nodes[0].CoreURL != "https://old.example" { t.Fatal("enrollment did not record the node's address", nodes, err) } s.SetPlacement(placementRules(t, "https://new.example")) - if _, err := s.CreateSession(t.Context(), uuid.NewString(), managerSessionInput(uuid.NewString())); !errors.Is(err, placement.ErrNodeUnavailable) { + w.SetPlacement(s.placement) + queued, err := s.CreateSession(t.Context(), uuid.NewString(), managerSessionInput(uuid.NewString())) + if err != nil { + t.Fatal(err) + } + if err := reserveSessionPlacement(t, s, w, queued); !errors.Is(err, placement.ErrNodeUnavailable) { t.Fatal("placed a new sandbox on a node bound to the old address", err) } bindings, err := deploymentService(t, s).AddressBindings(t.Context()) - if err != nil || bindings.Nodes != 1 || bindings.NodesOnOtherAddress != 1 || bindings.HostedSandboxes != 0 { + if err != nil || bindings.Nodes != 1 || bindings.NodesOnOtherAddress != 1 || bindings.HostedSandboxes != 1 { t.Fatal(bindings, err) } + var placements int + if err := s.pool.QueryRow(t.Context(), "SELECT count(*) FROM runtime_placements WHERE environment_id=$1", queued.Environment.ID).Scan(&placements); err != nil || placements != 0 { + t.Fatal("waiting Session acquired an incompatible placement", placements, err) + } s.SetPlacement(placementRules(t, "https://old.example")) - if _, err := s.CreateSession(t.Context(), uuid.NewString(), managerSessionInput(uuid.NewString())); err != nil { - t.Fatal("node on the current address rejected placement", err) + w.SetPlacement(s.placement) + if err := reserveSessionPlacement(t, s, w, queued); err != nil { + t.Fatal("node on current address rejected waiting placement", err) } } diff --git a/services/core/tests/integration/session_deletion_test.go b/services/core/tests/integration/session_deletion_test.go index f23b20593..f58924b7d 100644 --- a/services/core/tests/integration/session_deletion_test.go +++ b/services/core/tests/integration/session_deletion_test.go @@ -387,6 +387,9 @@ func TestSessionDeletionKeepsProvisioningInputPlacementUntilSettled(t *testing.T if err != nil { t.Fatal(err) } + if err := reserveSessionPlacement(t, s, w, session); err != nil { + t.Fatal(err) + } type state struct { deleted, released pgtype.Timestamptz retained, reserved int64 diff --git a/services/core/tests/integration/session_devices_fixture_test.go b/services/core/tests/integration/session_devices_fixture_test.go index 9dc9380e5..9181b28a3 100644 --- a/services/core/tests/integration/session_devices_fixture_test.go +++ b/services/core/tests/integration/session_devices_fixture_test.go @@ -20,6 +20,7 @@ func bindSessionDevice(t *testing.T, s *Store, tenant, session, device string) e if err != nil { return err } + defer func() { _ = lease.Close(context.Background()) }() released := pgtest.ObserveExecutionLeaseRelease(t, s.pool) operations, err := sessions.NewExecutionOperations(sessionpg.NewExecution(lease)) if err == nil { diff --git a/services/core/tests/integration/session_execution_configuration_test.go b/services/core/tests/integration/session_execution_configuration_test.go index e4a1c168c..3a047f2f8 100644 --- a/services/core/tests/integration/session_execution_configuration_test.go +++ b/services/core/tests/integration/session_execution_configuration_test.go @@ -272,6 +272,11 @@ func TestSessionExecutionConfigurationSurvivesSuspendResume(t *testing.T) { if phase == "running" { retained = nil } + if phase == "restoring" { + if err := deploymentService(t, s).TouchActivity(t.Context(), tenant, environment.ID); err != nil { + t.Fatal(err) + } + } owner = runtimeSuspensionStep(t, w, owner, phase, retained) before, err := deploymentStore(w).Activity(t.Context(), owner.ID) if err != nil { diff --git a/services/core/tools/e2b-provider/README.md b/services/core/tools/e2b-provider/README.md index 4a1f66722..48b6bad6a 100644 --- a/services/core/tools/e2b-provider/README.md +++ b/services/core/tools/e2b-provider/README.md @@ -82,4 +82,10 @@ The adapter implements the complete shared suspension group using opaque retaine Resume calls `Sandbox.connect` once with `on_resume=restore` and the existing native timeout. It preserves the native sandbox ID while advancing the shared logical generation. Fresh connection material is persisted before returning running. Unknown pause and connect outcomes are observed without replay; a missing connect receipt keeps execution unavailable. The adapter fences stale generations before commands and deletion. Consumed pause receipt cleanup preserves running compute. +Before native connect is admitted, recovery may durably close an exact restore operation under the allocation lock. Its `RestoreAttemptClosed` response binds the operation, target logical generation, same native ID and retained provenance; it never infers no dispatch from cloud absence. Closed IDs are kept for that retained generation (maximum 64, no eviction). A late fresh request for a closed ID is rejected. An additional closure at the bound fails closed; an already admitted or uncertain connect is never closed or replayed. The next retained generation replaces this history only after old retained requests can no longer match. + +Deleting an unconsumed settled pause checks the exact source incarnation and requires the final deletion candidates to be absent or the same paused native ID. It then confirms absence and consumes the receipt. Unknown deletion preserves the receipt and may settle after a lost reply or local write failure. Consumed cleanup still preserves a running resumed incarnation. The SDK has no atomic state-conditional deletion; manual cloud lifecycle changes outside the managed allocation lock are not supported. + The shared registration owns the 300-second idle duration and 86400-second retention default. Native timeout remains 3600 seconds and automatic native resume remains disabled. SDK calls use the configured official-compatible endpoint. Generated declarations and fixtures own the private helper shape. + +A no-intent suspension observation durably fences its OperationID before returning `SuspendSettled`. The private source-bound fence holds at most 64 closed attempts without eviction; a full journal fails closed. Rollback keeps the same generation and its fences. Only a strictly newer, exact owned compute incarnation replaces the fence, while stale-source requests fail ownership checks before native calls. diff --git a/services/core/tools/e2b-provider/helper_contract_generated.py b/services/core/tools/e2b-provider/helper_contract_generated.py index 7f420e0aa..d454cdd8c 100644 --- a/services/core/tools/e2b-provider/helper_contract_generated.py +++ b/services/core/tools/e2b-provider/helper_contract_generated.py @@ -17,7 +17,7 @@ REQUEST_FIELDS = ["Version","Operation","Config","Reference","References","Bootstrap","RuntimeBootstrap","Command","Compute","Suspend","Resume","Retained","Deadline"] RESPONSE_FIELDS = ["Version","State","Info","Command","ErrorCode","DeploymentValid","TemplateBuild","Templates","Builds","Observation"] RESUME_FIELDS = ["Workspace","Reference","OperationID","Retained","Target","ReconcileOnly"] -RETAINED_FIELDS = ["Reference","ID","Data","OperationID","SourceGeneration","SourceName","SourceID"] +RETAINED_FIELDS = ["Compatibility","Reference","ID","Data","OperationID","SourceGeneration","SourceName","SourceID"] SDK_VERSION = "2.51.0" SUSPEND_CONTROL_FILE = "/run/oac/daemon-suspend.json" SUSPEND_FIELDS = ["Reference","OperationID","Source","Retained","ReconcileOnly"] diff --git a/services/core/tools/e2b-provider/provider.py b/services/core/tools/e2b-provider/provider.py index 661a7635f..e224f6f47 100644 --- a/services/core/tools/e2b-provider/provider.py +++ b/services/core/tools/e2b-provider/provider.py @@ -352,10 +352,13 @@ def renew(self): Sandbox.set_timeout(cloud.sandbox_id, self.config['TimeoutSeconds'], **self.options()) return self.qualified(self.owns(Sandbox.get_info(cloud.sandbox_id, **self.options()))) - def kill(self): + def kill(self, expected_paused_id=None): if self.rejected_absence(): return found = self.discover() + if expected_paused_id is not None and (len(found) > 1 or any( + cloud.sandbox_id != expected_paused_id or cloud.state != 'paused' for cloud in found)): + raise Failure('unconfirmed') # The allocation flock excludes any still-running local Create helper. # A matching actual VM proves the original request reached allocation. known = bool((self.receipt.data or {}).get('ids')) diff --git a/services/core/tools/e2b-provider/suspension.py b/services/core/tools/e2b-provider/suspension.py index d37b860da..453fb9ac6 100644 --- a/services/core/tools/e2b-provider/suspension.py +++ b/services/core/tools/e2b-provider/suspension.py @@ -89,6 +89,15 @@ def suspend(self): q = self.request('Suspend') current = self.current() self.matches(q.get('Source'), current) + fence = self.record.get('suspend_fence') + closed = [] + if fence: + if fence['source'] == current: + closed = fence['closed'] + elif fence['source']['Generation'] >= current['Generation']: + raise Failure('ownership') + if q['OperationID'] in closed and not q['ReconcileOnly']: + raise Failure('ownership') handle = self.retained(q['OperationID'], current) if q.get('Retained') not in (None, handle): raise Failure('ownership') @@ -104,6 +113,13 @@ def suspend(self): cloud = self.cloud() if cloud.state != 'running': raise Failure('unconfirmed') + if q['OperationID'] not in closed: + if len(closed) >= 64: + raise Failure('unconfirmed') + closed = closed + [q['OperationID']] + # Close the exact attempt durably before Core may thaw this source. + # Rollback does not advance the generation and cannot clear fences. + self.p.receipt.save(suspend_fence={'source': current, 'closed': closed}) state = self.state(current, cloud) state['SuspendSettled'] = True return {'State': state} @@ -135,6 +151,23 @@ def resume(self): target = {'Generation': retained['SourceGeneration'] + 1, 'Name': retained['SourceName'], 'ID': retained['SourceID'], 'RestoredFrom': retained} self.matches(q.get('Target'), target) + closed = entry.get('closed_resume_ids', []) + if q['OperationID'] in closed: + if not q['ReconcileOnly'] or entry['phase'] != 'paused': + raise Failure('ownership') + return {'State': {'Compute': target, 'Status': 'absent', 'BootstrapComplete': False, + 'Retained': None, 'ResourcesReleased': False, 'SuspendSettled': False, + 'RestoreAttemptClosed': q['OperationID']}} + if entry['phase'] == 'paused' and q['ReconcileOnly']: + # Under the allocation flock no helper has persisted native dispatch + # for this attempt. Fence its late request durably before reporting it. + # Scope the bounded history to this retained generation; never evict. + if len(closed) >= 64: + raise Failure('unconfirmed') + self.p.receipt.save(suspension=dict(entry, closed_resume_ids=closed + [q['OperationID']])) + return {'State': {'Compute': target, 'Status': 'absent', 'BootstrapComplete': False, + 'Retained': None, 'ResourcesReleased': False, 'SuspendSettled': False, + 'RestoreAttemptClosed': q['OperationID']}} if entry['phase'] in ('resume_pending', 'resumed', 'consumed'): if entry.get('resume_id') != q['OperationID'] or entry.get('target') != target: raise Failure('ownership') @@ -207,6 +240,14 @@ def delete(self): raise Failure('ownership') if entry['phase'] == 'consumed': return {} + if self.record.get('status') != 'killed' and entry['phase'] == 'paused': + source = {'Generation': wanted['SourceGeneration'], 'Name': wanted['SourceName'], + 'ID': wanted['SourceID'], 'RestoredFrom': self.current()['RestoredFrom']} + self.matches(source, self.current()) + # A settled, unconsumed pause owns this exact native resource. The + # final discovery in kill rechecks paused ownership; unknown deletion + # keeps this receipt, and confirmed absence can settle a retry. + self.p.kill(expected_paused_id=source['ID']) if self.record.get('status') != 'killed': if entry['phase'] != 'resumed': raise Failure('unconfirmed') diff --git a/services/core/tools/e2b-provider/suspension_test.py b/services/core/tools/e2b-provider/suspension_test.py index 875c0d69e..de1d6c5e0 100644 --- a/services/core/tools/e2b-provider/suspension_test.py +++ b/services/core/tools/e2b-provider/suspension_test.py @@ -168,3 +168,121 @@ def killed(*args, **kwargs): self.api.connect.assert_called_once() self.api.kill.assert_called_once() self.assertEqual(self.record()['status'], 'killed') + def test_paused_delete_reclaims_same_native_id_and_settles_lost_reply(self): + from e2b.exceptions import SandboxNotFoundException + source = self.prepare() + _, paused = self.pause(source) + retained = paused['State']['Retained'] + def killed_then_lost(*args, **kwargs): + self.api.get_info.side_effect = SandboxNotFoundException('gone') + raise TimeoutError() + self.api.kill.side_effect = killed_then_lost + self.assertEqual(self.invoke('delete_retained', Retained=retained)['ErrorCode'], 'unconfirmed') + self.assertEqual(self.record()['suspension']['phase'], 'paused') + self.assertEqual(self.invoke('delete_retained', Retained=retained)['ErrorCode'], '') + self.assertEqual(self.record()['status'], 'killed') + self.assertEqual(self.record()['suspension']['phase'], 'consumed') + self.api.kill.assert_called_once() + self.assertEqual(self.api.kill.call_args.args[0], source['ID']) + + def test_paused_delete_rejects_running_source(self): + source = self.prepare() + _, paused = self.pause(source) + self.cloud.state = 'running' + self.assertEqual(self.invoke('delete_retained', Retained=paused['State']['Retained'])['ErrorCode'], 'unconfirmed') + self.assertEqual(self.record()['suspension']['phase'], 'paused') + self.api.kill.assert_not_called() + + def test_never_dispatched_restore_closes_durably_and_rejects_late_request(self): + source = self.prepare() + _, paused = self.pause(source) + q = self.resume_request(paused['State']['Retained']) + observed = self.invoke('resume', Resume=dict(q, ReconcileOnly=True)) + self.assertEqual(observed['ErrorCode'], '') + self.assertEqual(observed['State']['RestoreAttemptClosed'], q['OperationID']) + self.assertEqual(observed['State']['Compute'], q['Target']) + self.api.connect.assert_not_called() + self.assertEqual(self.invoke('resume', Resume=q)['ErrorCode'], 'ownership') + self.assertEqual(self.invoke('resume', Resume=dict(q, ReconcileOnly=True)), observed) + fresh = dict(q, OperationID=str(uuid4())) + self.assertEqual(self.invoke('resume', Resume=fresh)['ErrorCode'], '') + self.assertEqual(self.invoke('resume', Resume=dict(q, ReconcileOnly=True))['ErrorCode'], 'ownership') + self.api.connect.assert_called_once() + + def test_closed_restore_history_is_bounded_without_eviction(self): + source = self.prepare() + _, paused = self.pause(source) + requests = [] + for _ in range(64): + q = self.resume_request(paused['State']['Retained']) + requests.append(q) + self.assertEqual(self.invoke('resume', Resume=dict(q, ReconcileOnly=True))['ErrorCode'], '') + extra = self.resume_request(paused['State']['Retained']) + self.assertEqual(self.invoke('resume', Resume=dict(extra, ReconcileOnly=True))['ErrorCode'], 'unconfirmed') + self.assertEqual(len(self.record()['suspension']['closed_resume_ids']), 64) + self.assertEqual(self.invoke('resume', Resume=requests[0])['ErrorCode'], 'ownership') + self.api.connect.assert_not_called() + # A fresh, authorized attempt can still proceed; closing unknown attempts + # never consumes the original native retained image. + self.assertEqual(self.invoke('resume', Resume=extra)['ErrorCode'], '') + self.assertEqual(self.invoke('delete_retained', Retained=paused['State']['Retained'])['ErrorCode'], '') + current = self.record()['compute'] + _, next_pause = self.pause(current) + self.assertNotIn('closed_resume_ids', self.record()['suspension']) + self.assertEqual(self.invoke('resume', Resume=requests[0])['ErrorCode'], 'ownership') + + def test_paused_delete_settles_after_killed_receipt_write_fails(self): + from e2b.exceptions import SandboxNotFoundException + from state import Receipt + from unittest.mock import patch + source = self.prepare() + _, paused = self.pause(source) + retained = paused['State']['Retained'] + def killed(*args, **kwargs): + self.api.get_info.side_effect = SandboxNotFoundException('gone') + self.api.kill.side_effect = killed + save = Receipt.save + def fail_killed(receipt, **values): + if values.get('status') == 'killed': + raise OSError('fixture durable write failed') + return save(receipt, **values) + with patch.object(Receipt, 'save', fail_killed): + self.assertEqual(self.invoke('delete_retained', Retained=retained)['ErrorCode'], 'unconfirmed') + self.assertEqual(self.invoke('delete_retained', Retained=retained)['ErrorCode'], '') + self.api.kill.assert_called_once() + self.assertEqual(self.record()['suspension']['phase'], 'consumed') + + def test_rollback_closes_late_suspend_across_helper_reopen(self): + source = self.prepare() + q = {'Reference': self.reference, 'OperationID': str(uuid4()), 'Source': source, + 'Retained': None, 'ReconcileOnly': True} + result = self.invoke('suspend', Suspend=q) + self.assertTrue(result['State']['SuspendSettled']) + self.assertIsNone(result['State']['Retained']) + self.assertEqual(self.invoke('suspend', Suspend=dict(q, ReconcileOnly=False))['ErrorCode'], 'ownership') + self.assertEqual(self.cloud.state, 'running') + self.api.pause.assert_not_called() + self.assertEqual(self.invoke('suspend', Suspend=q), result) + + def test_suspend_fences_bound_and_survive_same_generation_rollback(self): + source = self.prepare() + requests = [] + for _ in range(64): + q = {'Reference': self.reference, 'OperationID': str(uuid4()), 'Source': source, + 'Retained': None, 'ReconcileOnly': True} + requests.append(q) + self.assertTrue(self.invoke('suspend', Suspend=q)['State']['SuspendSettled']) + extra = dict(q, OperationID=str(uuid4())) + self.assertEqual(self.invoke('suspend', Suspend=extra)['ErrorCode'], 'unconfirmed') + self.assertEqual(self.invoke('suspend', Suspend=dict(requests[0], ReconcileOnly=False))['ErrorCode'], 'ownership') + self.assertEqual(len(self.record()['suspend_fence']['closed']), 64) + # Only a real resume advances the incarnation; the next observation then + # replaces the journal while stale source requests remain fenced. + _, paused = self.pause(source) + resume = self.resume_request(paused['State']['Retained']) + current = self.invoke('resume', Resume=resume)['State']['Compute'] + self.assertEqual(self.invoke('delete_retained', Retained=paused['State']['Retained'])['ErrorCode'], '') + fresh = dict(extra, Source=current, OperationID=str(uuid4())) + self.assertTrue(self.invoke('suspend', Suspend=fresh)['State']['SuspendSettled']) + self.assertEqual(len(self.record()['suspend_fence']['closed']), 1) + self.assertEqual(self.invoke('suspend', Suspend=dict(requests[0], ReconcileOnly=False))['ErrorCode'], 'ownership') diff --git a/services/core/tools/microsandbox-provider/README.md b/services/core/tools/microsandbox-provider/README.md index a487b26f6..6db1c1822 100644 --- a/services/core/tools/microsandbox-provider/README.md +++ b/services/core/tools/microsandbox-provider/README.md @@ -1,26 +1,28 @@ # microsandbox Sandbox Provider helper -microsandbox runs each hosted Session in its own microVM on a Linux amd64 node with KVM, and it is the Sandbox Provider that supports idle suspension. Core forwards provider operations to the node over the [node protocol](../../../../contracts/agents-api/node-generation-protocol.md); the node's adapter ([`sandbox/microsandbox`](../../internal/sandbox/microsandbox)) runs this helper once per operation. The private helper wire version is 5; mismatches are rejected. The helper links the microsandbox Go SDK v0.7.2 with its FFI library, so Core and the node program stay CGO-free Go binaries. It implements the suspension operations of the provider-neutral `sandbox.SandboxProvider` with full snapshots. It has no daemon, lifecycle database, scheduler or network control plane. +microsandbox runs each hosted Session in its own microVM on a Linux amd64 node with KVM, and it is the Sandbox Provider that supports idle suspension. Core forwards provider operations to the node over the [node protocol](../../../../contracts/agents-api/node-generation-protocol.md); the node's adapter ([`sandbox/microsandbox`](../../internal/sandbox/microsandbox)) runs this helper once per operation. The private helper wire version is 5; mismatches are rejected. The helper links the microsandbox Go SDK v0.7.8 with its FFI library, so Core and the node program stay CGO-free Go binaries. It implements the suspension operations of the provider-neutral `sandbox.SandboxProvider` with full snapshots. It has no daemon, lifecycle database, scheduler or network control plane. [Add a Sandbox Provider](../../../../docs/sandbox-provider.md) owns the provider contract. [Sandbox deployment](../../../../contracts/agents-api/sandbox-deployment.md) owns the resources, Runtime release and suspension policy; the [nodes guide](../../../../docs/getting-started/nodes.md) owns node installation, host requirements, the node's directories and its network policy. ## Installation checks -The [maintainer guide](../../../../docs/maintainers.md#runtime-images-and-helpers) builds the helper and packages the checksum-verified `msb` runtime and `libkrunfw` firmware. The node's provider configuration, written by the node installer, supplies the absolute helper, runtime and firmware paths with their SHA-256 values, the runtime home, the Runtime image reference, the saved resources and the host network policy. +The [maintainer guide](../../../../docs/maintainers.md#runtime-images-and-helpers) builds the helper and packages the checksum-verified `msb` runtime and `libkrunfw` firmware. The node's provider configuration, written by the node installer, supplies the absolute helper, runtime and firmware paths with their SHA-256 values, the runtime home, the private checkpoint root, the Runtime image reference, the saved resources and the host network policy. -Before every operation the helper checks that it was built with the published SDK module v0.7.2 without a replacement, that the runtime and firmware match their hashes, that the runtime's `.msbver` ELF section reports the SDK version and that the SDK resolves exactly those paths with the local backend ([`main.go`](main.go)). It never installs or upgrades these files; keep them unchanged for the lifetime of the provider's backend. Ambient SDK profiles are ignored. +Before every operation the helper checks that it was built with the exact official SDK module declared in its embedded `go.mod` without a replacement, that the runtime and firmware match their hashes, that the runtime's `.msbver` ELF section reports the declared native release, that the SDK source and embedded FFI report that release, and that the SDK resolves exactly those paths with the local backend ([`main.go`](main.go)). It never installs or upgrades these files; keep them unchanged for the lifetime of the provider's backend. Ambient SDK profiles are ignored. The runtime home must be private (mode 0700), short, on local persistent storage and used by no other installation, profile or manual lifecycle tool. microsandbox uses Unix sockets there, so the node installer refuses a home whose path would exceed their limit. The home holds confidential VM disks, memory snapshots and SDK state; preserve it with the node identity and Core's database when recovering a host. Every managed lifecycle change goes through the provider. The Runtime image is an immutable `repository@sha256:<64 lowercase hex>` reference that matches the saved Runtime release; bare image IDs and mutable tags are rejected. The node installer imports the distribution's image under that reference. To load an image by hand, `msb image load --tag repository@sha256:` must register the digest reference explicitly, with the manifest digest from `image inspect`, not the Docker image config ID. The image carries the daemon, Python 3, the native Harnesses and the shared Runtime helpers. The provider installs no registry credentials. +The checkpoint root is a separate adapter-owned directory, mode 0700, with a private `.oac-checkpoint-store` UUID marker. It is never derived from a workspace binding and must not be reachable through any guest mount. Archives contain confidential RAM and root-disk state. The marker and installation identity define the artifact domain; native release and binary hashes, Runtime image, kernel, CPU features and guest geometry define the conservative execution class. Filesystem paths, hostnames and local SDK homes do not define compatibility. Nodes can share this private root at different mount paths; the backing store must preserve durable writes, atomic rename and cross-host file locks. The installer's local default only advertises its own domain. + ## Create and bootstrap -Create names the VM from a hash of the installation and allocation reference plus the compute generation; a name never serves another incarnation. It creates the VM with the saved CPUs and memory as both initial and maximum, a managed root disk of `root_disk_mib`, an owned ext4 disk of `environment_disk_mib` or the explicitly resolved external directory mounted at `/environment`, user 1000:1000, working directory `/` and the node's network policy ([`bootstrap.go`](bootstrap.go)). The resource checks run before bootstrap. +Create names the VM from a hash of the installation and allocation reference plus the compute generation; a name never serves another incarnation. It creates the VM with the saved CPUs and memory as both initial and maximum, a managed root disk of `root_disk_mib`, an owned ext4 disk of `environment_disk_mib` or the explicitly resolved external directory mounted at `/environment`, user 1000:1000, working directory `/` and the node's network policy ([`bootstrap.go`](bootstrap.go)). The resource checks run before bootstrap. Creation and restore retain the SDK's strict hostname policy enforcement; the adapter does not disable it. -Workspace, staging and outputs share the `/environment` filesystem, which keeps the Runtime's cross-device and link checks intact; the layered root filesystem can report different device IDs for a directory and its upper-layer files, so it holds no workspace data. In owned mode, full snapshots and sandbox removal capture, restore and reclaim this disk. In external mode, the filesystem provider owns the directory; sandbox removal never deletes it. The binding and resource rules belong to [Independent workspace attachment](../../../../docs/sandbox-provider.md#independent-workspace-attachment). Private HOME and history remain in the checkpointed root. +Workspace, staging and outputs share the `/environment` filesystem, which keeps the Runtime's cross-device and link checks intact; the layered root filesystem can report different device IDs for a directory and its upper-layer files, so it holds no workspace data. In owned mode, full snapshots and sandbox removal capture, restore and reclaim this disk. In external mode, the filesystem provider owns the directory; sandbox removal never deletes it. The binding and resource rules belong to [Independent workspace attachment](../../../../docs/sandbox-provider.md#independent-workspace-attachment). The [Runtime bootstrap](../../../../docs/runtime-bootstrap.md) owns the Runtime directory layout. -VM creation does not run the image's entry point. The bootstrap runs as root through confidential standard input, with a two-minute limit. It creates the Runtime directories and the private control directory `/run/oac` (mode 0700, owned by UID 1000), writes the [Runtime bootstrap](../../../../docs/runtime-bootstrap.md) file to `/home/runtime/runtime-bootstrap.json`, bind-mounts `/environment/workspace` at `/workspace` and starts `oac-daemon connect --bootstrap-file` in the background as UID/GID 1000. `OAC_RUNTIME_DAEMON_SUSPEND_PID_FILE=/run/oac/daemon-suspend.json` enables the daemon's park and wake control. The `io.oac.bootstrap` label then changes from `pending` to `complete` through the SDK's next-start modification policy, because v0.7.2 cannot update the labels of a running VM. That label confirms only these writes and the launch, not authentication or native readiness. +VM creation does not run the image's entry point. The bootstrap runs as root through confidential standard input, with a two-minute limit. It creates the Runtime directories and the private control directory `/run/oac` (mode 0700, owned by UID 1000), writes the [Runtime bootstrap](../../../../docs/runtime-bootstrap.md) file to `/home/runtime/runtime-bootstrap.json`, bind-mounts `/environment/workspace` at `/workspace` and starts `oac-daemon connect --bootstrap-file` in the background as UID/GID 1000. `OAC_RUNTIME_DAEMON_SUSPEND_PID_FILE=/run/oac/daemon-suspend.json` enables the daemon's park and wake control. The `io.oac.bootstrap` label then changes from `pending` to `complete` through the SDK's next-start modification policy, because v0.7.8 cannot update the labels of a running VM. That label confirms only these writes and the launch, not authentication or native readiness. A helper response carries `CreateSettled` with a configuration rejection only after native Create has completed, the first inspection has verified the exact compute ID and ownership, and the resource check has rejected the VM before bootstrap started. The adapter keeps the original error and validates the compute identity before passing the proof to Core. The private `initial_info` operation used by `GetInfo` and `Renew` can also return an exact initial `State` with `Status="absent"`, no native ID and `CreateSettled=true`, after typed native absence and durable closure of initial Create admission under the allocation lock. It accepts only generation zero without restored ancestry or a native ID. The adapter validates the complete receipt and returns absence without an error. Ordinary `inspect` used by `GetCompute`, uncertain Create outcomes, timeouts and ownership failures never produce this receipt; missing compute alone is not settlement. @@ -28,21 +30,23 @@ A helper response carries `CreateSettled` with a configuration rejection only af Core persists operation IDs, source and target generations, exact identities and snapshot evidence before it depends on them. `Initial` and `NewCompute` only construct references and allocate nothing. -- **Suspend** pauses the exact VM, captures a full snapshot under the persisted operation's derived group and member, verifies the complete checkpoint closure, then force-stops the source. Pausing alone does not release memory. A completed matching artifact is inspected instead of captured again. A full snapshot records a resource proof only after the source's limits match. -- **KillCompute** checks the precise incarnation before it stops the VM and removes its writable disks. Suspend completes native source cleanup under the allocation lock before reporting resource release, including recovery of a captured artifact. Core uses KillCompute for allocation cleanup. -- **Restore** verifies the exact artifact and creates the precommitted target name. An existing target is adopted only when its immutable ID, if known, and its persisted `snapshot_parent` agree. Upstream restore defaults to public networking, so restore passes the same explicit host policy as creation, and no undeclared host resource or mount is inherited. +- **Suspend** pauses the exact VM, captures and verifies the full checkpoint closure, then exports a standalone SDK archive with its image into the private checkpoint root. The adapter synchronizes the archive and its digest, size and ownership receipt before force-stopping the exact paused source, removing its writable disks and deleting its local SDK snapshot. Only after that cleanup is synchronized does `SourceStopped` authorize transfer. A matching artifact is completed instead of captured again; a running source is never killed to finish an old capture. +- **KillCompute** fences future admission for the exact target and checks its incarnation before stopping and removing it. Under that dispatch fence the SDK serializes against native launch, waits for the runtime lifecycle lock on Kill, and rechecks the terminal incarnation under that lock on Remove. A final typed absence permits a durable cleanup receipt, including for an admitted restore whose launcher died. This proves the writer is gone, never that it did not execute. +- **Restore** requires the source-settled archive receipt, matching artifact domain and execution class, and the complete archive hash before importing through the SDK and creating the precommitted target name. An existing target is adopted only when its immutable ID, if known, and its persisted `snapshot_parent` agree. Upstream restore defaults to public networking, so restore passes the same explicit host policy as creation, and no undeclared host resource or mount is inherited. - Native restore leaves the managed root size unset because the target inherits the verified full snapshot. The helper accepts that only with a matching snapshot resource proof and the exact source and target identities, and it checks the target's CPU, memory and Environment disk before keeping the inherited proof. A missing root size never counts as unlimited capacity, and retained state is never resized. - Fresh restore and retry share one completion: verify the original artifact and resource proof, inspect the running target's resources and ancestry, persist its missing derived resource-proof label, then strictly reread the same native ID ([`restore_completion.go`](restore_completion.go)). A conflicting proof is an error. Native restore does not copy the source's ownership labels; ancestry supplies that evidence. There is no ordinary Start, replacement, disk-only restore or cold boot. - **ResumeCompute** thaws the same resident source after an aborted suspension. The pinned SDK handle method is name-based; the allocation lock and ID checks before and after the call fence every managed replacement. Manual lifecycle changes in the managed namespace are unsupported. -- **DeleteRetained** accepts only the derived operation selector and the matching full artifact identity, never an arbitrary path. Core owns retention, consumed snapshot generations and cleanup order. A checkpoint never rolls back work admitted after its first restore. +- **DeleteRetained** accepts only the derived operation selector and matching artifact identity. It fences delayed publication and restoration with a durable deletion receipt before deleting the local SDK snapshot and portable archive. An unsettled admitted restore blocks deletion. The small ownership tombstone remains; Core owns retention and cleanup order. + +After a lost response Core uses `ObserveOnly`. It never starts a capture or restore. For an existing verified capture it may complete archive publication and exact paused-source cleanup, but never captures again or kills a running source. Source termination is recorded before local removal so a lost cleanup response can be settled from the archive receipt. A valid ownership receipt remains discoverable when archive bytes are missing or corrupt: observation returns its snapshot identity with unknown source state and `SourceStopped=false`, so ordinary explicit cleanup can proceed. Restore still requires complete archive verification. An interrupted restore may finish the derived proof on the exact running target, but never restarts a stopped target or restores again. -After a lost response Core uses `ReconcileOnly`. The adapter observes the original capture or restore and never starts it again. When a complete matching snapshot exists, reconciliation finishes exact source cleanup before reporting settled suspension and released resources. Artifact integrity and source ownership are checked independently of resource qualification, so resource drift cannot prevent owned cleanup. If capture is settled without an artifact and the exact source is still running or paused with a settled bootstrap, the adapter reports that outcome so Core can abort suspension; thawing and further execution still require resource checks. For an interrupted restore, reconciliation may finish the missing resource proof on the exact target but never restarts a stopped target, changes resources or restores again. Missing state never authorizes a replay. +Before native import or restore, the helper synchronizes an admission record bound to the request operation, snapshot and target. Observation of a never-admitted operation writes a permanent closed record and returns that exact `RestoreAttemptClosed`; late dispatch is rejected. A fresh dispatch checks its existing request deadline after admission and immediately before calling native Restore. If the deadline has elapsed, it durably closes its exact attempt while still holding the allocation lock and returns without invoking Restore. Native FFI waits remain uncancelled. Import or qualification errors keep the admission open, avoiding repeated attempts against an invalid archive. This proof is never inferred from a previous admission or a native Restore error. After admission, helper death or native absence alone never establishes no effect. Observation checks an admitted target before reading the archive; typed absence or a non-running target remains unconfirmed without hashing or importing the archive. A running target still requires complete archive verification. Successful resume requires an observable exact running target. A failed or uncertain admitted restore retains its artifact and owner until explicit deletion or the original retention deadline; the normal cleanup operation uses the native lifecycle fence and records completion. It never shortens retention or substitutes a cold restart. GetCompute, commands, cleanup and the next suspension verify restored provenance from the persisted VM configuration after the consumed artifact is deleted. ## Locks and commands -A helper holds a per-allocation lock, under `oac-locks/` in the runtime home, until its SDK call actually settles. Core's response deadline neither kills the helper nor cancels its FFI wait, because cancelling the wait does not prove that the native mutation stopped. On a timeout Core keeps an unknown operation and observes it; a later helper cannot pass the surviving lock holder. A stuck owner needs operator investigation, not lock deletion or another Create. +A helper holds a per-allocation lock in the explicitly configured checkpoint root, until its SDK call actually settles. Core's response deadline neither kills the helper nor cancels its FFI wait, because cancelling the wait does not prove that the native mutation stopped. On a timeout Core keeps an unknown operation and observes it; a later helper cannot pass the surviving lock holder. A stuck owner needs operator investigation, not lock deletion or another Create. The same lock file fences a Create helper that has not yet acquired its lock, including across a node restart. An empty file permits initial Create admission; `closed\n` permanently forbids it, and any other nonempty content fails closed. Only `initial_info` observing typed native absence writes this marker while holding the lock, then synchronizes the file and its directories before returning the receipt. Every Create checks the marker under the same lock before invoking the SDK. Read, write or synchronization errors never establish settlement. The marker is never removed, and neither missing compute nor this receipt permits retrying Create. @@ -57,7 +61,7 @@ The command checks the protected Environment and suspension identities and the p ## Metrics -The read-only metrics operation holds the allocation lock and verifies the exact compute ID through the SDK before and after it runs `msb metrics NAME --format json` with the same pinned runtime binary ([`metrics.go`](metrics.go)). The CLI report keeps the native sample timestamp and fractional-second uptime, so the helper reconstructs one run start consistently across polls and Core restarts, and a new run gets a new start. The Go SDK's projection drops the timestamp and truncates uptime to whole seconds, so it cannot provide this; sandbox creation time is not a run start time. In the pinned source (`v0.7.2`, commit `1c59b8dbf0ad47dda2f807c0214b529aceb81c74`), `crates/metrics/lib/registry.rs` reads `sampled_at_unix_ms` and `started_at_unix_ms` together and subtracts them for uptime, and `crates/cli/lib/commands/metrics.rs` serializes the timestamp and `uptime.as_secs_f64()`. +The read-only metrics operation holds the allocation lock and verifies the exact compute ID through the SDK before and after it runs `msb metrics NAME --format json` with the same pinned runtime binary ([`metrics.go`](metrics.go)). The CLI report keeps the native sample timestamp and fractional-second uptime, so the helper reconstructs one run start consistently across polls and Core restarts, and a new run gets a new start. The Go SDK's projection drops the timestamp and truncates uptime to whole seconds, so it cannot provide this; sandbox creation time is not a run start time. In the pinned source (`v0.7.8`, commit `7b7b9dc89e9a1c77801918f7833c3381d1754579`), `crates/metrics/lib/registry.rs` reads `sampled_at_unix_ms` and `started_at_unix_ms` together and subtracts them for uptime, and `crates/cli/lib/commands/metrics.rs` serializes the timestamp and `uptime.as_secs_f64()`. The helper returns only the native observation time, exact uptime, cumulative vCPU time and guest memory usage and limit. It rejects stale or exited reports and missing or malformed fields, bounds the CLI output and reports failures only as an unavailable code, never native diagnostics. Metrics never connect to the guest, renew activity, resume paused compute or change lifecycle state. [Runtime observability](../../../../contracts/agents-api/runtime-observability.md) owns the mapping to observations. @@ -66,3 +70,5 @@ The helper returns only the native observation time, exact uptime, cumulative vC `make check-microsandbox-provider`, part of `make check`, runs the pure-Go adapter tests everywhere and this module's tests on Linux; other hosts print an explicit skip for the Linux-only module. The native helper wire uses version 5 and carries the Session-selected Harness into Runtime bootstrap version 2. Upgrade the helper with its node and Core. The Go adapter wraps native snapshot identity in the shared opaque retained-state handle and completes exact-source cleanup through the existing ownership-checked helper operations. Reconciliation never repeats snapshot capture. SuspendSettled is reported only after these serialized operations complete; captured-artifact cleanup does not require execution resource qualification. + +Capture admission uses one adapter-private `capture-admission.json` per allocation, with the exact source and at most 64 operation entries. Before any pause/capture, a fresh call persists open admission. A no-archive observation closes only an operation with no admission; an open admission without an artifact stays unknown. Closure is durable before rollback evidence is returned and cannot be overwritten by a late fresh call. The source generation acts as a monotonic fence: only an exact owned newer source can replace the bounded journal; same-generation rollback cannot clear it. Existing verified archives and source-settlement receipts retain their original recovery path even after the source is stopped. diff --git a/services/core/tools/microsandbox-provider/archive.go b/services/core/tools/microsandbox-provider/archive.go new file mode 100644 index 000000000..2eba80326 --- /dev/null +++ b/services/core/tools/microsandbox-provider/archive.go @@ -0,0 +1,538 @@ +//go:build linux + +package main + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "encoding/json" + "errors" + "io" + "os" + "path/filepath" + "reflect" + "strings" + "time" + + "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" + wire "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" + sdk "github.com/superradcompany/microsandbox/sdk/go" +) + +// This adapter-private namespace is never a guest mount. The allocation lock +// serializes archive publication, admission and deletion across all its nodes. +type archiveReceipt struct { + Snapshot wire.SnapshotIdentity `json:"snapshot"` + SHA256 string `json:"sha256"` + Size int64 `json:"size"` + SourceTerminated bool `json:"source_terminated"` + SourceStopped bool `json:"source_stopped"` + Deleted bool `json:"deleted"` +} + +func (b backend) storeDirectory() string { + return filepath.Join(b.q.Config.CheckpointRoot, wire.Name(b.q.Config, b.q.Reference, 0)) +} +func (b backend) archiveDirectory(operation string) string { + return filepath.Join(b.storeDirectory(), "s-"+operation) +} +func privateDirectory(path string) error { + if err := os.Mkdir(path, 0700); err != nil && !errors.Is(err, os.ErrExist) { + return err + } + st, err := os.Lstat(path) + if err != nil { + return err + } + if !st.IsDir() || st.Mode().Perm() != 0700 { + return sandbox.ErrOwnership + } + return nil +} +func syncDirectory(path string) error { + f, e := os.Open(path) + if e != nil { + return e + } + defer f.Close() + return f.Sync() +} +func durableJSON(path string, value any) error { + data, e := json.Marshal(value) + if e != nil { + return e + } + f, e := os.CreateTemp(filepath.Dir(path), ".publish-") + if e != nil { + return e + } + defer os.Remove(f.Name()) + if _, e = f.Write(data); e == nil { + e = f.Sync() + } + closeErr := f.Close() + if e != nil { + return e + } + if closeErr != nil { + return closeErr + } + if e = os.Rename(f.Name(), path); e != nil { + return e + } + return syncDirectory(filepath.Dir(path)) +} +func readPrivateJSON(path string, value any) error { + st, e := os.Lstat(path) + if e != nil { + return e + } + if !st.Mode().IsRegular() || st.Mode().Perm() != 0600 { + return sandbox.ErrOwnership + } + real, e := filepath.EvalSymlinks(path) + if e != nil { + return e + } + if real != path { + return sandbox.ErrOwnership + } + f, e := os.Open(path) + if e != nil { + return e + } + defer f.Close() + d := json.NewDecoder(io.LimitReader(f, 65537)) + d.DisallowUnknownFields() + if e = d.Decode(value); e != nil { + return e + } + var extra any + if d.Decode(&extra) != io.EOF { + return sandbox.ErrOwnership + } + return nil +} +func (b backend) receipt(operation string) (archiveReceipt, error) { + var r archiveReceipt + e := readPrivateJSON(filepath.Join(b.archiveDirectory(operation), "receipt.json"), &r) + if e == nil && (r.Snapshot.OperationID != operation || wire.ValidateSnapshot(b.q.Config, b.q.Reference, r.Snapshot) != nil) { + e = sandbox.ErrOwnership + } + if e == nil && !r.Deleted { + decoded, err := hex.DecodeString(r.SHA256) + if err != nil || len(decoded) != 32 || r.Size <= 0 || (r.SourceStopped && !r.SourceTerminated) { + e = sandbox.ErrOwnership + } + } + return r, e +} +func archiveDigest(path string) (string, int64, error) { + real, e := filepath.EvalSymlinks(path) + if e != nil { + return "", 0, e + } + if real != path { + return "", 0, sandbox.ErrOwnership + } + st, e := os.Lstat(path) + if e != nil { + return "", 0, e + } + if !st.Mode().IsRegular() || st.Mode().Perm() != 0600 { + return "", 0, sandbox.ErrOwnership + } + f, e := os.Open(path) + if e != nil { + return "", 0, e + } + defer f.Close() + h := sha256.New() + n, e := io.Copy(h, f) + return hex.EncodeToString(h.Sum(nil)), n, e +} +func (b backend) publishArchive(ctx context.Context, a *sdk.SnapshotArtifact, s wire.SnapshotIdentity) (archiveReceipt, error) { + prior, e := b.receipt(s.OperationID) + if e == nil { + if prior.Snapshot != s || prior.Deleted { + return prior, sandbox.ErrOwnership + } + return prior, nil + } + if !errors.Is(e, os.ErrNotExist) { + return prior, e + } + dir := b.archiveDirectory(s.OperationID) + if e = privateDirectory(dir); e != nil { + return prior, e + } + if e = syncDirectory(b.storeDirectory()); e != nil { + return prior, e + } + archive := filepath.Join(dir, "archive.msb") + // A publication interrupted before its receipt can be rebuilt from the exact + // verified native artifact. The shared allocation lock excludes consumers. + if e = os.Remove(archive); e != nil && !errors.Is(e, os.ErrNotExist) { + return prior, e + } + if e = a.SaveTo(ctx, archive, sdk.SnapshotSaveOptions{WithImage: true}); e != nil { + return prior, e + } + if e = os.Chmod(archive, 0600); e != nil { + return prior, e + } + f, e := os.Open(archive) + if e != nil { + return prior, e + } + e = f.Sync() + f.Close() + if e != nil { + return prior, e + } + digest, size, e := archiveDigest(archive) + if e != nil { + return prior, e + } + r := archiveReceipt{Snapshot: s, SHA256: digest, Size: size} + return r, durableJSON(filepath.Join(dir, "receipt.json"), r) +} +func (b backend) importArchive(ctx context.Context, want wire.SnapshotIdentity) (*sdk.SnapshotArtifact, error) { + r, e := b.receipt(want.OperationID) + if e != nil { + return nil, e + } + if r.Deleted || !r.SourceStopped || r.Snapshot != want { + return nil, sandbox.ErrOwnership + } + class, e := wire.CheckpointClass(b.q.Config) + if e != nil { + return nil, e + } + if class != want.Compatibility { + return nil, sandbox.ErrInvalid + } + path := filepath.Join(b.archiveDirectory(want.OperationID), "archive.msb") + hash, size, e := archiveDigest(path) + if e != nil { + return nil, e + } + if hash != r.SHA256 || size != r.Size { + return nil, sandbox.ErrOwnership + } + artifact, e := b.verifiedSnapshot(ctx, want) + if sdk.IsKind(e, sdk.ErrSnapshotNotFound) { + if _, e = sdk.Snapshot.LoadWithOptions(ctx, path, sdk.SnapshotLoadOptions{Group: wire.Name(b.q.Config, b.q.Reference, 0)}); e != nil { + return nil, e + } + return b.verifiedSnapshot(ctx, want) + } + return artifact, e +} +func (b backend) deleteArchive(ctx context.Context, want wire.SnapshotIdentity) error { + return b.deleteArchiveWithNative(want, func() error { + _, err := b.verifiedSnapshot(ctx, want) + if err == nil { + err = sdk.Snapshot.Remove(ctx, want.Reference, false) + } + if sdk.IsKind(err, sdk.ErrSnapshotNotFound) { + return nil + } + return err + }) +} + +// Native artifact removal must settle before the private archive is unlinked. +func (b backend) deleteArchiveWithNative(want wire.SnapshotIdentity, removeNative func() error) error { + if e := b.requireSettledRestores(nil, &want); e != nil { + return e + } + r, e := b.receipt(want.OperationID) + if errors.Is(e, os.ErrNotExist) { + r = archiveReceipt{Snapshot: want} + if e = privateDirectory(b.archiveDirectory(want.OperationID)); e != nil { + return e + } + if e = syncDirectory(b.storeDirectory()); e != nil { + return e + } + } else if e != nil { + return e + } + if r.Snapshot != want { + return sandbox.ErrOwnership + } + // Fence consumers before removing either local or portable artifact. Keep the + // small tombstone, including its ownership identity, against delayed helpers. + r.Deleted = true + if e = durableJSON(filepath.Join(b.archiveDirectory(want.OperationID), "receipt.json"), r); e != nil { + return e + } + if e = removeNative(); e != nil { + return e + } + if e = os.Remove(filepath.Join(b.archiveDirectory(want.OperationID), "archive.msb")); e != nil && !errors.Is(e, os.ErrNotExist) { + return e + } + return syncDirectory(b.archiveDirectory(want.OperationID)) +} + +type restoreAdmission struct { + OperationID string + Snapshot wire.SnapshotIdentity + Target wire.Compute + Closed bool + CompletedID string + Cleaned bool +} + +func (b backend) admitRestore(q wire.ResumeRequest) (bool, error) { + if _, e := os.Lstat(filepath.Join(b.storeDirectory(), "k-"+q.Target.Name+".json")); e == nil { + return false, wire.ErrUnconfirmed + } else if !errors.Is(e, os.ErrNotExist) { + return false, e + } + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + want := restoreAdmission{OperationID: q.OperationID, Snapshot: q.Snapshot, Target: q.Target} + want.Target.ID = "" + var prior restoreAdmission + e := readPrivateJSON(path, &prior) + if e == nil { + closed := prior.Closed + prior.Closed = false + prior.CompletedID = "" + cleaned := prior.Cleaned + prior.Cleaned = false + if !reflect.DeepEqual(prior, want) { + return false, sandbox.ErrOwnership + } + if closed { + return true, nil + } + if cleaned { + return false, wire.ErrUnconfirmed + } + if !q.ObserveOnly { + return false, wire.ErrUnconfirmed + } + return false, nil + } + if !errors.Is(e, os.ErrNotExist) { + return false, e + } + if q.Target.ID != "" { + return false, sandbox.ErrOwnership + } + want.Closed = q.ObserveOnly + if e = durableJSON(path, want); e != nil { + return false, e + } + return want.Closed, nil +} + +// Only the current fresh dispatch, under the allocation lock and before its +// first RestoreSandbox call, may record this proof. A recovered open admission +// or native restore error cannot establish that no guest execution occurred. +func (b backend) closeExpiredRestore(q wire.ResumeRequest) error { + if time.Now().Before(b.q.Deadline) { + return nil + } + if q.ObserveOnly || q.Target.ID != "" { + return sandbox.ErrOwnership + } + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + var prior restoreAdmission + if err := readPrivateJSON(path, &prior); err != nil { + return err + } + want := restoreAdmission{OperationID: q.OperationID, Snapshot: q.Snapshot, Target: q.Target} + if !reflect.DeepEqual(prior, want) { + return sandbox.ErrOwnership + } + want.Closed = true + if err := durableJSON(path, want); err != nil { + return err + } + return context.DeadlineExceeded +} + +// Completion records the incarnation whose native restore has become observable +// and whose full provenance and running state have been verified. This is +// execution evidence; cleanup separately fences dispatch and uses native locks. +func (b backend) completeRestore(ctx context.Context, q wire.ResumeRequest, target wire.Compute) (wire.State, error) { + state, e := b.finishRestore(ctx, target) + if e != nil { + return state, e + } + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + var record restoreAdmission + if e = readPrivateJSON(path, &record); e != nil { + return state, e + } + if record.Closed || record.Snapshot != q.Snapshot || record.Target.Name != target.Name || record.Target.Generation != target.Generation { + return state, sandbox.ErrOwnership + } + if record.CompletedID != "" && record.CompletedID != state.Compute.ID { + return state, sandbox.ErrOwnership + } + record.CompletedID = state.Compute.ID + return state, durableJSON(path, record) +} +func (b backend) requireSettledRestores(target *wire.Compute, snapshot *wire.SnapshotIdentity) error { + files, e := os.ReadDir(b.storeDirectory()) + if e != nil { + return e + } + for _, entry := range files { + if !strings.HasPrefix(entry.Name(), "r-") || !strings.HasSuffix(entry.Name(), ".json") { + continue + } + var r restoreAdmission + if e = readPrivateJSON(filepath.Join(b.storeDirectory(), entry.Name()), &r); e != nil { + return e + } + if entry.Name() != "r-"+r.OperationID+".json" { + return sandbox.ErrOwnership + } + if r.Closed || r.Cleaned { + continue + } + if target != nil && (r.Target.Name != target.Name || r.Target.Generation != target.Generation) { + continue + } + if snapshot != nil && r.Snapshot != *snapshot { + continue + } + if r.CompletedID == "" { + return wire.ErrUnconfirmed + } + if target != nil && target.ID == "" { + target.ID = r.CompletedID + } + if target != nil && target.ID != "" && target.ID != r.CompletedID { + return sandbox.ErrOwnership + } + } + return nil +} + +func (b backend) verifyArchive(r archiveReceipt) error { + hash, size, e := archiveDigest(filepath.Join(b.archiveDirectory(r.Snapshot.OperationID), "archive.msb")) + if e != nil { + return e + } + if hash != r.SHA256 || size != r.Size { + return sandbox.ErrOwnership + } + return nil +} + +// The native transition/lifecycle locks fence an admitted launcher and its +// detached child. This receipt means cleanup completed, never "did not run". +func (b backend) recordRestoreCleanup(target wire.Compute) error { + files, e := os.ReadDir(b.storeDirectory()) + if e != nil { + return e + } + for _, entry := range files { + if !strings.HasPrefix(entry.Name(), "r-") || !strings.HasSuffix(entry.Name(), ".json") { + continue + } + path := filepath.Join(b.storeDirectory(), entry.Name()) + var r restoreAdmission + if e = readPrivateJSON(path, &r); e != nil { + return e + } + if entry.Name() != "r-"+r.OperationID+".json" { + return sandbox.ErrOwnership + } + if r.Target.Name != target.Name || r.Target.Generation != target.Generation || r.Closed || r.Cleaned { + continue + } + if target.ID != "" && r.CompletedID != "" && target.ID != r.CompletedID { + return sandbox.ErrOwnership + } + r.Cleaned = true + if e = durableJSON(path, r); e != nil { + return e + } + } + return nil +} + +func (b backend) pinRestoreCleanup(target *wire.Compute) error { + files, e := os.ReadDir(b.storeDirectory()) + if e != nil { + return e + } + for _, entry := range files { + if !strings.HasPrefix(entry.Name(), "r-") || !strings.HasSuffix(entry.Name(), ".json") { + continue + } + var r restoreAdmission + if e = readPrivateJSON(filepath.Join(b.storeDirectory(), entry.Name()), &r); e != nil { + return e + } + if entry.Name() != "r-"+r.OperationID+".json" { + return sandbox.ErrOwnership + } + if r.Target.Name != target.Name || r.Target.Generation != target.Generation || r.Closed { + continue + } + if target.RestoredFrom == nil || *target.RestoredFrom != r.Snapshot { + return sandbox.ErrOwnership + } + if r.CompletedID != "" { + if target.ID != "" && target.ID != r.CompletedID { + return sandbox.ErrOwnership + } + target.ID = r.CompletedID + } + } + return nil +} + +// One bounded journal fences capture admission across process restarts. Source +// generations only advance after suspend verified the new exact native source. +// Keeping that high-water identity rejects delayed requests from older writers. +type captureAdmission struct { + Source wire.Compute + Operations map[string]bool // true means durably closed without dispatch +} + +func (b backend) admitSuspend(q wire.SuspendRequest) (bool, error) { + path := filepath.Join(b.storeDirectory(), "capture-admission.json") + var prior captureAdmission + err := readPrivateJSON(path, &prior) + if err != nil && !errors.Is(err, os.ErrNotExist) { + return false, err + } + if err == nil { + if prior.Source.ID == "" || prior.Operations == nil || len(prior.Operations) > 64 { + return false, sandbox.ErrOwnership + } + if !reflect.DeepEqual(prior.Source, q.Source) { + if q.Source.Generation <= prior.Source.Generation { + return false, sandbox.ErrOwnership + } + prior = captureAdmission{Source: q.Source, Operations: map[string]bool{}} + } + } else { + prior = captureAdmission{Source: q.Source, Operations: map[string]bool{}} + } + if closed, exists := prior.Operations[q.OperationID]; exists { + if !closed && !q.ObserveOnly { + return false, wire.ErrUnconfirmed + } + return closed, nil + } + if len(prior.Operations) >= 64 { + return false, wire.ErrUnconfirmed + } + prior.Operations[q.OperationID] = q.ObserveOnly + if err := durableJSON(path, prior); err != nil { + return false, err + } + return q.ObserveOnly, nil +} diff --git a/services/core/tools/microsandbox-provider/backend.go b/services/core/tools/microsandbox-provider/backend.go index a9e6d9d55..cc27005c6 100644 --- a/services/core/tools/microsandbox-provider/backend.go +++ b/services/core/tools/microsandbox-provider/backend.go @@ -5,6 +5,7 @@ package main import ( "context" "errors" + "path/filepath" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" wire "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" @@ -51,14 +52,7 @@ func (b backend) run(ctx context.Context) (wire.Response, error) { s, e := b.resume(ctx, *b.q.Resume) return wire.Response{State: &s}, e case "delete_snapshot": - _, e := b.verifiedSnapshot(ctx, *b.q.Snapshot) - if sdk.IsKind(e, sdk.ErrSnapshotNotFound) { - return wire.Response{}, nil - } - if e != nil { - return wire.Response{}, e - } - return wire.Response{}, sdk.Snapshot.Remove(ctx, b.q.Snapshot.Reference, false) + return wire.Response{}, b.deleteArchive(ctx, *b.q.Snapshot) } return wire.Response{}, sandbox.ErrInvalid } @@ -81,13 +75,23 @@ func (b backend) inspectOwned(ctx context.Context, c wire.Compute) (*sdk.Sandbox return h, state, e } func (b backend) kill(ctx context.Context, c wire.Compute) error { + // Close dispatch before consulting native state. An admitted helper has either + // released this allocation flock or died; the SDK's transition and inherited + // lifecycle locks fence any surviving native launcher/runtime. + if e := durableJSON(filepath.Join(b.storeDirectory(), "k-"+c.Name+".json"), c); e != nil { + return e + } + if e := b.pinRestoreCleanup(&c); e != nil { + return e + } h, _, e := b.inspectOwned(ctx, c) if sdk.IsKind(e, sdk.ErrSandboxNotFound) { - return nil + return b.recordRestoreCleanup(c) } if e != nil { return e } + c.ID = h.ID() if e = h.Kill(ctx); e != nil { return e } @@ -96,13 +100,14 @@ func (b backend) kill(ctx context.Context, c wire.Compute) error { } _, e = sdk.GetSandbox(ctx, c.Name) if sdk.IsKind(e, sdk.ErrSandboxNotFound) { - return nil + return b.recordRestoreCleanup(c) } if e == nil { return errors.New("compute removal unconfirmed") } return e } + func (b backend) network() *sdk.NetworkConfig { n := b.q.Config.Network result := &sdk.NetworkConfig{DefaultEgress: sdk.PolicyAction(n.DefaultEgress), DefaultIngress: sdk.PolicyAction(n.DefaultIngress)} diff --git a/services/core/tools/microsandbox-provider/bootstrap.go b/services/core/tools/microsandbox-provider/bootstrap.go index e6779414f..d7911a82c 100644 --- a/services/core/tools/microsandbox-provider/bootstrap.go +++ b/services/core/tools/microsandbox-provider/bootstrap.go @@ -108,7 +108,7 @@ func (b backend) create(ctx context.Context) (wire.Response, error) { return wire.Response{}, sandbox.ErrCommandUnconfirmed } // Persist the final bootstrap receipt without restarting the live guest. - // v0.7.2 cannot update active labels; ownership reads persisted config. + // v0.7.8 cannot update active labels; ownership reads persisted config. _, e = live.Modify(ctx, sdk.ModifyOptions{Labels: map[string]string{bootstrapLabel: "complete"}, Policy: sdk.ModificationPolicyNextStart}) if e != nil { return wire.Response{}, e diff --git a/services/core/tools/microsandbox-provider/deployment.go b/services/core/tools/microsandbox-provider/deployment.go index c950588aa..c768e0e59 100644 --- a/services/core/tools/microsandbox-provider/deployment.go +++ b/services/core/tools/microsandbox-provider/deployment.go @@ -31,7 +31,7 @@ func resourceProof(config wire.Config) string { return hex.EncodeToString(digest[:]) } -// The SDK decodes native CPU/memory/rootfs configuration. Its v0.7.2 projection +// The SDK decodes native CPU/memory/rootfs configuration. Its v0.7.8 projection // omits mounts, so the explicitly owned or external inventory is projected here. // Restored roots have no configured size: a verified full snapshot supplies that // proof, retained as a label only after inspecting the restored target. diff --git a/services/core/tools/microsandbox-provider/go.mod b/services/core/tools/microsandbox-provider/go.mod index 25d0423c5..c04558e14 100644 --- a/services/core/tools/microsandbox-provider/go.mod +++ b/services/core/tools/microsandbox-provider/go.mod @@ -4,7 +4,7 @@ go 1.26.8 require ( github.com/MiniMax-AI/OpenAgentCore v0.0.0 - github.com/superradcompany/microsandbox/sdk/go v0.7.2 + github.com/superradcompany/microsandbox/sdk/go v0.0.0-20261009155513-7b7b9dc89e9a ) require ( diff --git a/services/core/tools/microsandbox-provider/go.sum b/services/core/tools/microsandbox-provider/go.sum index 77cb4e04c..a92270ad0 100644 --- a/services/core/tools/microsandbox-provider/go.sum +++ b/services/core/tools/microsandbox-provider/go.sum @@ -1,6 +1,6 @@ github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= -github.com/superradcompany/microsandbox/sdk/go v0.7.2 h1:aO70srBRgu50rNZTtzQTOO3xxPSthkZLkq3kjWvmsqg= -github.com/superradcompany/microsandbox/sdk/go v0.7.2/go.mod h1:p7Tm/p9zkO7sHO3N8A1fiTVeLB2w/fQoKdYlpN/5neA= +github.com/superradcompany/microsandbox/sdk/go v0.0.0-20261009155513-7b7b9dc89e9a h1:r9mZpz3z3E5tXRWDulZSs0gyM2EEhsTUqUjyZQ6MT4w= +github.com/superradcompany/microsandbox/sdk/go v0.0.0-20261009155513-7b7b9dc89e9a/go.mod h1:p7Tm/p9zkO7sHO3N8A1fiTVeLB2w/fQoKdYlpN/5neA= golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek= golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0= diff --git a/services/core/tools/microsandbox-provider/lock_test.go b/services/core/tools/microsandbox-provider/lock_test.go index 46c93595f..02e0ab5eb 100644 --- a/services/core/tools/microsandbox-provider/lock_test.go +++ b/services/core/tools/microsandbox-provider/lock_test.go @@ -16,8 +16,8 @@ import ( ) func TestAllocationLockSurvivesCallerDeadlineUntilExplicitSettlement(t *testing.T) { - home := t.TempDir() - q := wire.Request{Config: wire.Config{RuntimeHome: home}, Deadline: time.Now().Add(time.Second)} + home := checkpointFixture(t) + q := wire.Request{Config: wire.Config{RuntimeHome: home, CheckpointRoot: home}, Deadline: time.Now().Add(time.Second)} release, e := allocationLock(q) if e != nil { t.Fatal(e) @@ -39,12 +39,12 @@ func TestAllocationLockSurvivesCallerDeadlineUntilExplicitSettlement(t *testing. release.Close() } func TestLockDirectoryCannotRedirectIntoAnotherHome(t *testing.T) { - home := t.TempDir() + home := checkpointFixture(t) foreign := t.TempDir() - if e := os.Symlink(foreign, filepath.Join(home, "oac-locks")); e != nil { + if e := os.Symlink(foreign, filepath.Join(home, wire.Name(wire.Config{}, sandbox.Reference{}, 0))); e != nil { t.Fatal(e) } - _, e := allocationLock(wire.Request{Config: wire.Config{RuntimeHome: home}, Deadline: time.Now().Add(time.Second)}) + _, e := allocationLock(wire.Request{Config: wire.Config{RuntimeHome: home, CheckpointRoot: home}, Deadline: time.Now().Add(time.Second)}) if e == nil { t.Fatal("symlink lock directory accepted") } @@ -56,6 +56,7 @@ func initialInfoRequest(t *testing.T) wire.Request { config.InstallationID = "11111111-1111-4111-8111-111111111111" config.HelperPath, config.RuntimePath, config.FirmwarePath = "/helper", "/runtime", "/firmware" config.RuntimeHome = t.TempDir() + config.CheckpointRoot = checkpointFixture(t) config.RuntimeSHA256, config.FirmwareSHA256 = strings.Repeat("a", 64), strings.Repeat("b", 64) config.Network = wire.NetworkPolicy{DefaultEgress: "deny", DefaultIngress: "deny"} ref := sandbox.Reference{TenantID: "22222222-2222-4222-8222-222222222222", EnvironmentID: "33333333-3333-4333-8333-333333333333", AllocationID: "44444444-4444-4444-8444-444444444444"} @@ -193,3 +194,15 @@ func TestInvalidOrUnwritableAdmissionStateFailsClosed(t *testing.T) { } }) } + +func checkpointFixture(t *testing.T) string { + t.Helper() + root := t.TempDir() + if e := os.Chmod(root, 0700); e != nil { + t.Fatal(e) + } + if e := os.WriteFile(filepath.Join(root, ".oac-checkpoint-store"), []byte("55555555-5555-4555-8555-555555555555\n"), 0600); e != nil { + t.Fatal(e) + } + return root +} diff --git a/services/core/tools/microsandbox-provider/main.go b/services/core/tools/microsandbox-provider/main.go index c16a7d8da..f6220baae 100644 --- a/services/core/tools/microsandbox-provider/main.go +++ b/services/core/tools/microsandbox-provider/main.go @@ -8,6 +8,7 @@ import ( "context" "crypto/sha256" "debug/elf" + _ "embed" "encoding/hex" "encoding/json" "errors" @@ -25,6 +26,9 @@ import ( sdk "github.com/superradcompany/microsandbox/sdk/go" ) +//go:embed go.mod +var moduleDefinition string + func main() { // The inherited node generation lease covers this helper's actual lifetime. // Set CLOEXEC before SDK initialization or any possible subprocess spawn so @@ -54,6 +58,13 @@ func serve(input io.Reader) wire.Response { out.ErrorCode = "unconfirmed" return out } + if q.Workspace != nil { + real, err := filepath.EvalSymlinks(q.Workspace.Path) + if err != nil || real != q.Workspace.Path { + out.ErrorCode = "ownership" + return out + } + } lock, err := allocationLock(q) if err != nil { out.ErrorCode = "unconfirmed" @@ -169,7 +180,10 @@ func (g *allocationGuard) settleInitialAbsence(q wire.Request, nativeErr error) } func allocationLock(q wire.Request) (*allocationGuard, error) { - dir := filepath.Join(q.Config.RuntimeHome, "oac-locks") + if _, e := wire.CheckpointStore(q.Config); e != nil { + return nil, e + } + dir := filepath.Join(q.Config.CheckpointRoot, wire.Name(q.Config, q.Reference, 0)) if e := os.MkdirAll(dir, 0700); e != nil { return nil, e } @@ -180,6 +194,9 @@ func allocationLock(q wire.Request) (*allocationGuard, error) { if !st.IsDir() || st.Mode().Perm() != 0700 { return nil, sandbox.ErrOwnership } + if e := syncDirectory(q.Config.CheckpointRoot); e != nil { + return nil, e + } name := filepath.Join(dir, wire.Name(q.Config, q.Reference, 0)+".lock") fd, e := syscall.Open(name, syscall.O_CREAT|syscall.O_RDWR|syscall.O_NOFOLLOW|syscall.O_CLOEXEC, 0600) if e != nil { @@ -207,6 +224,16 @@ func allocationLock(q wire.Request) (*allocationGuard, error) { time.Sleep(20 * time.Millisecond) } } +func sdkModuleVersion() string { + for _, line := range strings.Split(moduleDefinition, "\n") { + fields := strings.Fields(line) + if len(fields) == 2 && fields[0] == "github.com/superradcompany/microsandbox/sdk/go" { + return fields[1] + } + } + return "" +} + func checkInstallation(c wire.Config) error { build, ok := debug.ReadBuildInfo() if !ok { @@ -215,7 +242,7 @@ func checkInstallation(c wire.Config) error { matched := false for _, d := range build.Deps { if d.Path == "github.com/superradcompany/microsandbox/sdk/go" { - matched = d.Version == wire.SDKVersion && d.Replace == nil + matched = d.Version == sdkModuleVersion() && d.Replace == nil } } if !matched { @@ -269,6 +296,13 @@ func checkInstallation(c wire.Config) error { if runtime.MSBPath != c.RuntimePath || runtime.LibkrunfwPath != c.FirmwarePath { return sandbox.ErrOwnership } + nativeVersion, e := sdk.RuntimeVersion() + if e != nil { + return e + } + if sdk.SDKVersion() != strings.TrimPrefix(wire.SDKVersion, "v") || nativeVersion != sdk.SDKVersion() { + return sandbox.ErrInvalid + } selected, e := sdk.DefaultBackendInfo() if e != nil { return e diff --git a/services/core/tools/microsandbox-provider/restore_completion_test.go b/services/core/tools/microsandbox-provider/restore_completion_test.go index 2dbd5ec55..5f0b13a14 100644 --- a/services/core/tools/microsandbox-provider/restore_completion_test.go +++ b/services/core/tools/microsandbox-provider/restore_completion_test.go @@ -61,7 +61,7 @@ func TestRestoreCompletionRecoversOnlyDerivedProof(t *testing.T) { } func TestRestoreCompletionRejectsUnverifiedTargets(t *testing.T) { - for _, fault := range []string{"cpu", "memory", "environment", "image", "root", "foreign parent", "foreign ID", "foreign proof", "stopped", "unfinished", "replacement after write", "proof not persisted"} { + for _, fault := range []string{"cpu", "memory", "environment", "image", "root", "foreign parent", "foreign ID", "foreign proof", "stopped", "starting", "paused", "crashed", "unfinished", "replacement after write", "proof not persisted"} { t.Run(fault, func(t *testing.T) { config, actual := deploymentFixture() actual["image"].(map[string]any)["Oci"].(map[string]any)["root_disk"].(map[string]any)["size_mib"] = nil @@ -87,8 +87,8 @@ func TestRestoreCompletionRejectsUnverifiedTargets(t *testing.T) { actualID = "local:foreign" case "foreign proof": labels[resourceProofLabel] = "foreign" - case "stopped": - status = "stopped" + case "stopped", "starting", "paused", "crashed": + status = fault case "unfinished": actual["checkpoint_restore"] = map[string]string{"checkpoint_id": "pending"} } diff --git a/services/core/tools/microsandbox-provider/sdk_release_test.go b/services/core/tools/microsandbox-provider/sdk_release_test.go new file mode 100644 index 000000000..7d528949f --- /dev/null +++ b/services/core/tools/microsandbox-provider/sdk_release_test.go @@ -0,0 +1,28 @@ +//go:build linux + +package main + +import ( + "strings" + "testing" + + wire "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" + sdk "github.com/superradcompany/microsandbox/sdk/go" +) + +func TestOfficialSDKReleaseMatchesEmbeddedFFI(t *testing.T) { + t.Setenv("MSB_HOME", t.TempDir()) + if sdkModuleVersion() == "" { + t.Fatal("missing pinned SDK module identity") + } + if sdk.SDKVersion() != strings.TrimPrefix(wire.SDKVersion, "v") { + t.Fatal("SDK source does not match admitted native release") + } + version, err := sdk.RuntimeVersion() + if err != nil { + t.Fatal(err) + } + if version != sdk.SDKVersion() { + t.Fatalf("embedded FFI %q does not match SDK %q", version, sdk.SDKVersion()) + } +} diff --git a/services/core/tools/microsandbox-provider/snapshot.go b/services/core/tools/microsandbox-provider/snapshot.go index 6cb6ee539..36ee2e4e4 100644 --- a/services/core/tools/microsandbox-provider/snapshot.go +++ b/services/core/tools/microsandbox-provider/snapshot.go @@ -5,7 +5,10 @@ package main import ( "context" "encoding/json" + "errors" "fmt" + "os" + "path/filepath" "strconv" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" @@ -49,7 +52,8 @@ func (b backend) inspectSnapshot(ctx context.Context, operation string, source w if report.Digest != artifact.Digest() || report.Checkpoint == nil || report.Checkpoint.Root == "" || report.Checkpoint.Kind == "" || state.Checkpoint == nil || state.Checkpoint.CheckpointID == "" { return nil, wire.SnapshotIdentity{}, sandbox.ErrOwnership } - identity := wire.SnapshotIdentity{Reference: selector, ID: artifact.ID(), Digest: artifact.Digest(), CheckpointID: state.Checkpoint.CheckpointID, CheckpointRoot: report.Checkpoint.Root, OperationID: operation, SourceName: source.Name, SourceID: source.ID, SourceGeneration: source.Generation} + compatibility := sandbox.CheckpointCompatibility{ArtifactDomain: artifact.Labels()["io.oac.artifact_domain"], ExecutionClass: artifact.Labels()["io.oac.execution_class"]} + identity := wire.SnapshotIdentity{Compatibility: compatibility, Reference: selector, ID: artifact.ID(), Digest: artifact.Digest(), CheckpointID: state.Checkpoint.CheckpointID, CheckpointRoot: report.Checkpoint.Root, OperationID: operation, SourceName: source.Name, SourceID: source.ID, SourceGeneration: source.Generation} if wire.ValidateSnapshot(b.q.Config, b.q.Reference, identity) != nil { return nil, wire.SnapshotIdentity{}, sandbox.ErrOwnership } @@ -66,11 +70,41 @@ func (b backend) verifiedSnapshot(ctx context.Context, want wire.SnapshotIdentit return a, nil } func (b backend) suspend(ctx context.Context, q wire.SuspendRequest) (wire.State, error) { + receipt, receiptErr := b.receipt(q.OperationID) + if receiptErr == nil { + if receipt.Deleted || receipt.Snapshot.SourceName != q.Source.Name || receipt.Snapshot.SourceID != q.Source.ID || receipt.Snapshot.SourceGeneration != q.Source.Generation || (q.Snapshot != nil && *q.Snapshot != receipt.Snapshot) { + return wire.State{}, sandbox.ErrOwnership + } + return b.settleSource(ctx, q, receipt) + } else if !errors.Is(receiptErr, os.ErrNotExist) { + return wire.State{}, receiptErr + } + // The host allocation flock spans pause, capture, verification and source kill. // A surviving completed artifact is observed, never overwritten or recaptured. artifact, snap, e := b.inspectSnapshot(ctx, q.OperationID, q.Source) if sdk.IsKind(e, sdk.ErrSnapshotNotFound) { + // Native absence cannot settle an admitted capture. Only a durable + // never-dispatched closure permits rollback of this exact source. + // Existing archives/receipts above retain their source-cleanup path. + _, sourceState, inspectErr := b.inspectOwned(ctx, q.Source) + if inspectErr != nil { + return wire.State{}, inspectErr + } + if !sourceState.BootstrapComplete || (sourceState.Status != "running" && sourceState.Status != "paused") { + return wire.State{}, wire.ErrUnconfirmed + } + closed, admissionErr := b.admitSuspend(q) + if admissionErr != nil { + return wire.State{}, admissionErr + } + if !q.ObserveOnly && closed { + return wire.State{}, wire.ErrUnconfirmed + } if q.ObserveOnly { + if !closed { + return wire.State{}, wire.ErrUnconfirmed + } if q.Snapshot != nil { return wire.State{}, wire.ErrUnconfirmed } @@ -97,6 +131,12 @@ func (b backend) suspend(ctx context.Context, q wire.SuspendRequest) (wire.State return wire.State{}, err } labels := b.snapshotLabels(q.OperationID, q.Source) + compatibility, err := wire.CheckpointClass(b.q.Config) + if err != nil { + return wire.State{}, err + } + labels["io.oac.artifact_domain"] = compatibility.ArtifactDomain + labels["io.oac.execution_class"] = compatibility.ExecutionClass var captured struct { Labels map[string]string `json:"labels"` } @@ -120,23 +160,37 @@ func (b backend) suspend(ctx context.Context, q wire.SuspendRequest) (wire.State if q.Snapshot != nil && snap != *q.Snapshot { return wire.State{}, sandbox.ErrOwnership } - if q.ObserveOnly { - // A completed operation receipt remains observable for cleanup even if - // its source or resource proof is no longer qualified for execution. - return observeCapturedSnapshot(q.Source, snap, func(source wire.Compute) (wire.State, error) { - _, state, err := b.inspectOwned(ctx, source) - return state, err - }) - } if e = qualifySnapshotResources(b.q.Config, artifact.Labels()); e != nil { return wire.State{}, e } - h, _, e := b.inspect(ctx, q.Source) + receipt, e = b.publishArchive(ctx, artifact, snap) + if e != nil { + return wire.State{}, e + } + return b.settleSource(ctx, q, receipt) +} + +func (b backend) settleSource(ctx context.Context, q wire.SuspendRequest, r archiveReceipt) (wire.State, error) { + if e := b.verifyArchive(r); e != nil { + if q.ObserveOnly { + // Ownership comes from the durable receipt, independently of archive + // usability. Expose cleanup identity without authorizing source transfer, + // claiming its current state, or weakening the restore integrity check. + return wire.State{Compute: q.Source, Status: "unknown", Snapshot: &r.Snapshot}, nil + } + return wire.State{}, e + } + if r.SourceStopped { + return wire.State{Compute: q.Source, Status: "suspended", BootstrapComplete: true, Snapshot: &r.Snapshot, SourceStopped: true}, nil + } + h, state, e := b.inspectOwned(ctx, q.Source) if e != nil && !sdk.IsKind(e, sdk.ErrSandboxNotFound) { return wire.State{}, e } - if e == nil { - // Graceful stop could run the captured source after the checkpoint. + if e == nil && !terminal(sdk.SandboxStatus(state.Status)) { + if state.Status != "paused" { + return wire.State{}, wire.ErrUnconfirmed + } if e = h.Kill(ctx); e != nil { return wire.State{}, e } @@ -147,12 +201,84 @@ func (b backend) suspend(ctx context.Context, q wire.SuspendRequest) (wire.State if !terminal(observed.Status()) { return wire.State{}, fmt.Errorf("source termination unconfirmed") } + } else if sdk.IsKind(e, sdk.ErrSandboxNotFound) && !r.SourceTerminated { + return wire.State{}, wire.ErrUnconfirmed + } + if !r.SourceTerminated { + r.SourceTerminated = true + if e = durableJSON(filepath.Join(b.archiveDirectory(q.OperationID), "receipt.json"), r); e != nil { + return wire.State{}, e + } + } + if h != nil { + if e = h.Remove(ctx); e != nil { + return wire.State{}, e + } + } + if _, e = sdk.GetSandbox(ctx, q.Source.Name); !sdk.IsKind(e, sdk.ErrSandboxNotFound) { + return wire.State{}, wire.ErrUnconfirmed + } + if e = sdk.Snapshot.Remove(ctx, r.Snapshot.Reference, false); e != nil && !sdk.IsKind(e, sdk.ErrSnapshotNotFound) { + return wire.State{}, e } - return wire.State{Compute: q.Source, Status: "suspended", BootstrapComplete: true, Snapshot: &snap, SourceStopped: true}, nil + r.SourceStopped = true + if e = durableJSON(filepath.Join(b.archiveDirectory(q.OperationID), "receipt.json"), r); e != nil { + return wire.State{}, e + } + return wire.State{Compute: q.Source, Status: "suspended", BootstrapComplete: true, Snapshot: &r.Snapshot, SourceStopped: true}, nil } + func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, error) { - // Verify the exact full artifact before either adopting or creating a target. - artifact, e := b.verifiedSnapshot(ctx, q.Snapshot) + // No managed target may predate its exact admission journal. + if _, err := os.Lstat(filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json")); errors.Is(err, os.ErrNotExist) { + _, err = sdk.GetSandbox(ctx, q.Target.Name) + if err == nil { + return wire.State{}, sandbox.ErrOwnership + } + if !sdk.IsKind(err, sdk.ErrSandboxNotFound) { + return wire.State{}, err + } + } else if err != nil { + return wire.State{}, err + } + + closed, e := b.admitRestore(q) + if e != nil { + return wire.State{}, e + } + if closed { + if !q.ObserveOnly { + return wire.State{}, wire.ErrUnconfirmed + } + target := q.Target + target.ID = "" + return wire.State{Compute: target, Status: "absent", RestoreAttemptClosed: q.OperationID}, nil + } + // An existing admitted observation may cheaply remain unknown. Native absence + // cannot close it: a detached launcher could still make the target visible. + if q.ObserveOnly { + _, observed, err := b.inspectOwned(ctx, q.Target) + if sdk.IsKind(err, sdk.ErrSandboxNotFound) { + return wire.State{}, wire.ErrUnconfirmed + } + if err != nil { + return wire.State{}, err + } + if observed.Status != "running" { + return wire.State{}, wire.ErrUnconfirmed + } + } + // Only a fresh dispatch may close its attempt before native execution. + // Lifecycle FFI calls use an uncancelled context; the request deadline is + // checked between settled operations, never used to cancel a native wait. + if !q.ObserveOnly { + if err := b.closeExpiredRestore(q); err != nil { + return wire.State{}, err + } + } + // Admission is durable before any native import/restore operation. A prior + // admitted operation with no visible target remains unknown, never replayed. + artifact, e := b.importArchive(ctx, q.Snapshot) if e != nil { return wire.State{}, e } @@ -164,7 +290,7 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, } _, _, e = b.inspectOwned(ctx, q.Target) if e == nil { - return b.finishRestore(ctx, q.Target) + return b.completeRestore(ctx, q, q.Target) } if !sdk.IsKind(e, sdk.ErrSandboxNotFound) { return wire.State{}, e @@ -172,21 +298,13 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, if q.ObserveOnly || q.Target.ID != "" { return wire.State{}, wire.ErrUnconfirmed } - source, err := sdk.GetSandbox(ctx, q.Snapshot.SourceName) - if err == nil { - if source.ID() != q.Snapshot.SourceID { - return wire.State{}, sandbox.ErrOwnership - } - if !terminal(source.Status()) { - return wire.State{}, wire.ErrUnconfirmed - } - } else if !sdk.IsKind(err, sdk.ErrSandboxNotFound) { - return wire.State{}, err - } restore := sdk.RestoreConfig{NetworkPolicy: b.network(), ExternalMountPolicy: sdk.ExternalMountStrict} if b.q.Workspace != nil { restore.Volumes = map[string]sdk.MountConfig{"/environment": sdk.Mount.Bind(b.q.Workspace.Path, sdk.MountOptions{})} } + if err := b.closeExpiredRestore(q); err != nil { + return wire.State{}, err + } live, e := sdk.RestoreSandbox(ctx, artifact, q.Target.Name, sdk.WithRestoreConfig(restore)) if e != nil { return wire.State{}, e @@ -195,7 +313,7 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, target.ID = live.ID() // The same completion path handles a fresh target and a previous Restore // whose response or derived proof write was interrupted. - state, err := b.finishRestore(ctx, target) + state, err := b.completeRestore(ctx, q, target) detachErr := live.Detach(context.Background()) if err != nil { return state, err @@ -206,21 +324,6 @@ func (b backend) resume(ctx context.Context, q wire.ResumeRequest) (wire.State, return state, nil } -// Observation consumes an already verified artifact identity and never changes -// source state. Execution qualification belongs to capture and subsequent use. -func observeCapturedSnapshot(source wire.Compute, snapshot wire.SnapshotIdentity, readOwned func(wire.Compute) (wire.State, error)) (wire.State, error) { - state, err := readOwned(source) - if err != nil && !sdk.IsKind(err, sdk.ErrSandboxNotFound) { - return wire.State{}, err - } - stopped := sdk.IsKind(err, sdk.ErrSandboxNotFound) || terminal(sdk.SandboxStatus(state.Status)) - status := "suspended" - if !stopped { - status = state.Status - } - return wire.State{Compute: source, Status: status, BootstrapComplete: true, Snapshot: &snapshot, SourceStopped: stopped}, nil -} - func terminal(s sdk.SandboxStatus) bool { return s == sdk.SandboxStatusStopped || s == sdk.SandboxStatusCrashed } diff --git a/services/core/tools/microsandbox-provider/snapshot_observation_test.go b/services/core/tools/microsandbox-provider/snapshot_observation_test.go index cf2a80b07..7ebd62a6d 100644 --- a/services/core/tools/microsandbox-provider/snapshot_observation_test.go +++ b/services/core/tools/microsandbox-provider/snapshot_observation_test.go @@ -3,83 +3,587 @@ package main import ( - "encoding/json" + "context" "errors" + "fmt" + "os" + "os/exec" + "path/filepath" "testing" + "time" "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox" wire "github.com/MiniMax-AI/OpenAgentCore/services/core/internal/sandbox/microsandbox" - sdk "github.com/superradcompany/microsandbox/sdk/go" ) -func TestCapturedSnapshotObservationDoesNotRequireExecutionQualification(t *testing.T) { - for _, drift := range []string{"cpu", "image", "missing snapshot proof", "foreign snapshot proof"} { - t.Run(drift, func(t *testing.T) { - config, actual := deploymentFixture() - ref := sandbox.Reference{TenantID: "tenant", EnvironmentID: "environment", AllocationID: "allocation"} - source := wire.Compute{Name: "original", ID: "local:original"} - labels := wire.Labels(config, ref) - labels[workspaceModeLabel] = "owned" - labels[bootstrapLabel] = "complete" - actual["labels"] = labels - snapshotLabels := map[string]string{workspaceModeLabel: "owned", resourceProofLabel: resourceProof(config)} - switch drift { - case "cpu": - actual["resources"].(map[string]any)["cpus"] = 1 - case "image": - actual["image"] = "runtime:drifted" - case "missing snapshot proof": - delete(snapshotLabels, resourceProofLabel) - case "foreign snapshot proof": - snapshotLabels[resourceProofLabel] = "foreign" - } - raw, _ := json.Marshal(actual) - if qualifyConfiguration(config, source, string(raw), false) == nil && qualifySnapshotResources(config, snapshotLabels) == nil { - t.Fatal("fixture did not disqualify execution") - } - // inspectSnapshot has already verified the exact artifact and closure. - snapshot := wire.SnapshotIdentity{ID: "verified", SourceID: source.ID, SourceName: source.Name} - reads := 0 - state, err := observeCapturedSnapshot(source, snapshot, func(want wire.Compute) (wire.State, error) { - reads++ - return qualifyCompute(config, ref, want, source.ID, "paused", string(raw)) - }) - if err != nil || state.Snapshot == nil || *state.Snapshot != snapshot || state.Compute != source || state.SourceStopped || reads != 1 { - t.Fatal("owned receipt blocked by execution drift", state, err, reads) +func TestRestoreObservationClosesOnlyNeverAdmittedOperation(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "exact-target", Generation: 1}, ObserveOnly: true} + closed, err := b.admitRestore(q) + if err != nil || !closed { + t.Fatal(closed, err) + } + q.ObserveOnly = false + // A late helper cannot replace the durable closed record with admission. + closed, err = b.admitRestore(q) + if err != nil || !closed { + t.Fatal("late restore admitted", closed, err) + } + q.OperationID = "77777777-7777-4777-8777-777777777777" + closed, err = b.admitRestore(q) + if err != nil || closed { + t.Fatal(closed, err) + } + // A crash after admission remains unknown, including when no target is yet + // visible. The helper never manufactures no-effect evidence from absence. + q.ObserveOnly = true + closed, err = b.admitRestore(q) + if err != nil || closed { + t.Fatal("admitted operation closed", closed, err) + } + q.ObserveOnly = false + if _, err = b.admitRestore(q); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("admitted operation replayed", err) + } + q.ObserveOnly = true + q.Target.Name = "foreign" + if _, err = b.admitRestore(q); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal("foreign target admitted", err) + } +} +func TestRestoreJournalRejectsPartialAndSymlinkRecords(t *testing.T) { + for _, kind := range []string{"partial", "symlink"} { + t.Run(kind, func(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", ObserveOnly: true} + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + if kind == "partial" { + err = os.WriteFile(path, []byte("{"), 0600) + } else { + err = os.Symlink(filepath.Join(t.TempDir(), "missing"), path) + } + if err != nil { + t.Fatal(err) + } + if _, err = b.admitRestore(q); err == nil { + t.Fatal("unsafe record accepted") } }) } } +func TestArchiveDigestRejectsGuestReadableAndSymlinkArtifacts(t *testing.T) { + dir := t.TempDir() + path := filepath.Join(dir, "archive") + if err := os.WriteFile(path, []byte("private RAM"), 0600); err != nil { + t.Fatal(err) + } + hash, size, err := archiveDigest(path) + if err != nil || len(hash) != 64 || size != 11 { + t.Fatal(hash, size, err) + } + if err = os.Chmod(path, 0644); err != nil { + t.Fatal(err) + } + if _, _, err = archiveDigest(path); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal(err) + } + link := filepath.Join(dir, "link") + if err = os.Symlink(path, link); err != nil { + t.Fatal(err) + } + if _, _, err = archiveDigest(link); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal(err) + } +} -func TestCapturedSnapshotObservationRetainsSourceOwnershipAndState(t *testing.T) { - source := wire.Compute{Name: "original", ID: "local:original"} - snapshot := wire.SnapshotIdentity{ID: "verified", SourceID: source.ID} - for _, test := range []struct { - name, status string - err error - stopped bool - }{ - {name: "running", status: "running"}, - {name: "unknown", status: "unknown"}, - {name: "stopped", status: "stopped", stopped: true}, - {name: "crashed", status: "crashed", stopped: true}, - {name: "absent", err: &sdk.Error{Kind: sdk.ErrSandboxNotFound}, stopped: true}, - {name: "foreign", err: sandbox.ErrOwnership}, - {name: "unavailable", err: wire.ErrUnconfirmed}, - } { - t.Run(test.name, func(t *testing.T) { - state, err := observeCapturedSnapshot(source, snapshot, func(wire.Compute) (wire.State, error) { - return wire.State{Compute: source, Status: test.status}, test.err +func TestAdmittedRestoreBlocksCleanupUntilExactCompletion(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "exact-target", Generation: 1}} + if _, err = b.admitRestore(q); err != nil { + t.Fatal(err) + } + if err = b.requireSettledRestores(&q.Target, nil); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("unknown target was releasable", err) + } + if err = b.requireSettledRestores(nil, &q.Snapshot); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("unknown artifact was releasable", err) + } + other := wire.Compute{Name: "other-source"} + if err = b.requireSettledRestores(&other, nil); err != nil { + t.Fatal("unrelated source cleanup blocked", err) + } + record := restoreAdmission{OperationID: q.OperationID, Target: q.Target, Snapshot: q.Snapshot, CompletedID: "local:exact"} + if err = durableJSON(filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json"), record); err != nil { + t.Fatal(err) + } + if err = b.requireSettledRestores(&q.Target, nil); err != nil || q.Target.ID != "local:exact" { + t.Fatal("cleanup did not pin completed incarnation", q.Target, err) + } + q.Target.ID = "local:foreign" + if err = b.requireSettledRestores(&q.Target, nil); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal("foreign cleanup accepted", err) + } +} + +func TestTargetCleanupFenceSurvivesReopenBeforeAndAfterAdmission(t *testing.T) { + for _, admitted := range []bool{false, true} { + t.Run(fmt.Sprint(admitted), func(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "exact-target", Generation: 1}} + if admitted { + if _, err = b.admitRestore(q); err != nil { + t.Fatal(err) + } + } + if err = durableJSON(filepath.Join(b.storeDirectory(), "k-"+q.Target.Name+".json"), q.Target); err != nil { + t.Fatal(err) + } + guard.Close() + guard, err = allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + if _, err = b.admitRestore(q); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("late restore crossed cleanup fence", err) + } + if admitted { + // Only the caller which has completed native Get/Kill/Remove/Get invokes + // this journal update. The live crash probe qualifies the native fence. + if err = b.recordRestoreCleanup(q.Target); err != nil { + t.Fatal(err) + } + if err = b.requireSettledRestores(nil, &q.Snapshot); err != nil { + t.Fatal("cleaned obligation retained", err) + } + } + }) + } +} + +func TestLostCaptureResponseStillExposesOwnedIdentityWhenArchiveIsDamaged(t *testing.T) { + for _, settled := range []bool{false, true} { + for _, damage := range []string{"corrupt", "missing"} { + t.Run(fmt.Sprintf("settled=%t/%s", settled, damage), func(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + source := request.Compute + source.ID = "local:known-source" + operation := "66666666-6666-4666-8666-666666666666" + compatibility, err := wire.CheckpointClass(request.Config) + if err != nil { + t.Fatal(err) + } + snapshot := wire.SnapshotIdentity{Compatibility: compatibility, Reference: wire.SnapshotReference(request.Config, request.Reference, operation), ID: "exact-snapshot", Digest: "digest", CheckpointID: "checkpoint", CheckpointRoot: "root", OperationID: operation, SourceName: source.Name, SourceID: source.ID} + dir := b.archiveDirectory(operation) + if err = privateDirectory(dir); err != nil { + t.Fatal(err) + } + archive := filepath.Join(dir, "archive.msb") + if err = os.WriteFile(archive, []byte("verified full archive"), 0600); err != nil { + t.Fatal(err) + } + hash, size, err := archiveDigest(archive) + if err != nil { + t.Fatal(err) + } + receipt := archiveReceipt{Snapshot: snapshot, SHA256: hash, Size: size, SourceTerminated: settled, SourceStopped: settled} + if err = durableJSON(filepath.Join(dir, "receipt.json"), receipt); err != nil { + t.Fatal(err) + } + if damage == "corrupt" { + err = os.WriteFile(archive, []byte("damaged"), 0600) + } else { + err = os.Remove(archive) + } + if err != nil { + t.Fatal(err) + } + // Core lost the capture response and therefore has no Snapshot yet. It + // needs this ownership identity to reach ordinary Kill/DeleteSnapshot. + state, err := b.suspend(t.Context(), wire.SuspendRequest{Reference: request.Reference, OperationID: operation, Source: source, ObserveOnly: true}) + if err != nil || state.Snapshot == nil || *state.Snapshot != snapshot || state.Compute != source || state.SourceStopped || state.Status != "unknown" || state.BootstrapComplete { + t.Fatalf("cleanup identity hidden or execution authorized: %+v %v", state, err) + } + if artifact, err := b.importArchive(t.Context(), snapshot); err == nil || artifact != nil { + t.Fatal("damaged archive was usable for restore", err) + } + if _, err = b.suspend(t.Context(), wire.SuspendRequest{Reference: request.Reference, OperationID: operation, Source: source}); err == nil { + t.Fatal("non-observation accepted unusable archive") + } + foreign := source + foreign.ID = "local:foreign" + denied, err := b.suspend(t.Context(), wire.SuspendRequest{Reference: request.Reference, OperationID: operation, Source: foreign, ObserveOnly: true}) + if !errors.Is(err, sandbox.ErrOwnership) || denied.Snapshot != nil { + t.Fatal("foreign request recovered owned identity", denied, err) + } + nativeRemoved := false + if err = b.deleteArchiveWithNative(*state.Snapshot, func() error { nativeRemoved = true; return nil }); err != nil { + t.Fatal("owned damaged archive could not be deleted", err) + } + if !nativeRemoved { + t.Fatal("native cleanup was skipped") + } + if _, err = os.Lstat(archive); !errors.Is(err, os.ErrNotExist) { + t.Fatal("archive remains", err) + } + deleted, err := b.receipt(operation) + if err != nil || !deleted.Deleted || deleted.Snapshot != snapshot { + t.Fatal("exact deletion fence missing", deleted, err) + } }) - if test.err != nil && !test.stopped { - if !errors.Is(err, test.err) || state.Snapshot != nil { - t.Fatal("unverified source returned a successful observation", state, err) + } + } +} + +// These requests use the real SDK's empty local registry, without starting a VM. +func TestPermanentRestoreImportFailureKeepsOneAdmission(t *testing.T) { + if !isolatedSDKTest(t) { + return + } + home := t.TempDir() + t.Setenv("MSB_HOME", home) + for _, damage := range []string{"missing", "corrupt"} { + t.Run(damage, func(t *testing.T) { + request := initialInfoRequest(t) + request.Config.RuntimeHome = home + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: wire.Name(request.Config, request.Reference, 1), Generation: 1}} + if damage == "corrupt" { + compatibility, err := wire.CheckpointClass(request.Config) + if err != nil { + t.Fatal(err) + } + q.Snapshot = wire.SnapshotIdentity{Compatibility: compatibility, Reference: wire.SnapshotReference(request.Config, request.Reference, q.OperationID), ID: "id", Digest: "digest", CheckpointID: "checkpoint", CheckpointRoot: "root", OperationID: q.OperationID, SourceName: wire.Name(request.Config, request.Reference, 0), SourceID: "local:source"} + dir := b.archiveDirectory(q.OperationID) + if err = privateDirectory(dir); err != nil { + t.Fatal(err) + } + archive := filepath.Join(dir, "archive.msb") + if err = os.WriteFile(archive, []byte("original"), 0600); err != nil { + t.Fatal(err) + } + hash, size, err := archiveDigest(archive) + if err != nil { + t.Fatal(err) + } + if err = durableJSON(filepath.Join(dir, "receipt.json"), archiveReceipt{Snapshot: q.Snapshot, SHA256: hash, Size: size, SourceStopped: true, SourceTerminated: true}); err != nil { + t.Fatal(err) } - return + if err = os.WriteFile(archive, []byte("corrupt!"), 0600); err != nil { + t.Fatal(err) + } + } + if _, err = b.resume(t.Context(), q); err == nil { + t.Fatal("missing archive accepted") + } + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + var record restoreAdmission + if err = readPrivateJSON(path, &record); err != nil || record.Closed { + t.Fatal("permanent failure closed", record, err) + } + before, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + files, err := os.ReadDir(b.storeDirectory()) + if err != nil { + t.Fatal(err) + } + q.ObserveOnly = true + for i := 0; i < 3; i++ { + state, err := b.resume(t.Context(), q) + if !errors.Is(err, wire.ErrUnconfirmed) || state.RestoreAttemptClosed != "" { + t.Fatal(state, err) + } + } + after, err := os.ReadFile(path) + if err != nil || string(before) != string(after) { + t.Fatal("journal changed", err) + } + afterFiles, err := os.ReadDir(b.storeDirectory()) + if err != nil || len(afterFiles) != len(files) { + t.Fatal("journal grew", err) + } + }) + } +} + +func TestFreshRestoreRequestDeadlineClosesBeforeNativeExecution(t *testing.T) { + if !isolatedSDKTest(t) { + return + } + request := initialInfoRequest(t) + t.Setenv("MSB_HOME", request.Config.RuntimeHome) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + // The real request acquired its lock before its deadline. Time can expire + // while prior SDK work settles; the lifecycle context deliberately stays live. + request.Deadline = time.Now().Add(-time.Second) + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: wire.Name(request.Config, request.Reference, 1), Generation: 1}} + request.Operation = "resume" + request.Resume = &q + b := backend{q: request} + _, err = b.run(context.Background()) + if !errors.Is(err, context.DeadlineExceeded) { + t.Fatal("request deadline not enforced", err) + } + q.ObserveOnly = true + state, err := b.resume(context.Background(), q) + if err != nil || state.RestoreAttemptClosed != q.OperationID || state.Status != "absent" { + t.Fatal("closed proof missing", state, err) + } + q.ObserveOnly = false + if _, err = b.resume(context.Background(), q); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("late request replayed", err) + } +} + +// The official SDK freezes its default backend on first use. Match production's +// one-request helper process rather than changing MSB_HOME under a cached pool. +func isolatedSDKTest(t *testing.T) bool { + t.Helper() + if os.Getenv("OAC_NATIVE_TEST_CHILD") == t.Name() { + return true + } + cmd := exec.Command(os.Args[0], "-test.run=^"+t.Name()+"$", "-test.count=1") + cmd.Env = append(os.Environ(), "OAC_NATIVE_TEST_CHILD="+t.Name()) + if output, err := cmd.CombinedOutput(); err != nil { + t.Fatalf("isolated SDK test: %v\n%s", err, output) + } + return false +} + +func TestAdmittedAbsentObservationSkipsArchiveAndStaysUnknown(t *testing.T) { + if !isolatedSDKTest(t) { + return + } + request := initialInfoRequest(t) + t.Setenv("MSB_HOME", request.Config.RuntimeHome) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: wire.Name(request.Config, request.Reference, 1), Generation: 1}} + if _, err = b.admitRestore(q); err != nil { + t.Fatal(err) + } + // Even a missing archive must not be consulted: its ENOENT would differ from + // the typed unknown result returned for this existing admitted target. + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + before, err := os.ReadFile(path) + if err != nil { + t.Fatal(err) + } + b.q.Deadline = time.Now().Add(-time.Second) + q.ObserveOnly = true + state, err := b.resume(t.Context(), q) + if !errors.Is(err, wire.ErrUnconfirmed) || state.RestoreAttemptClosed != "" { + t.Fatal(state, err) + } + after, err := os.ReadFile(path) + if err != nil || string(before) != string(after) { + t.Fatal("observation mutated admission", err) + } + q.ObserveOnly = false + if _, err = b.resume(t.Context(), q); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("old admission replayed", err) + } +} + +func TestPreDispatchClosureRejectsConflictingJournal(t *testing.T) { + for _, kind := range []string{"observe", "completed", "cleaned", "foreign", "missing"} { + t.Run(kind, func(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + request.Deadline = time.Now().Add(-time.Second) + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "target", Generation: 1}} + r := restoreAdmission{OperationID: q.OperationID, Target: q.Target, Snapshot: q.Snapshot} + switch kind { + case "observe": + q.ObserveOnly = true + case "completed": + r.CompletedID = "local:1" + case "cleaned": + r.Cleaned = true + case "foreign": + r.Target.Name = "foreign" } - if err != nil || state.Snapshot == nil || state.SourceStopped != test.stopped { - t.Fatal("observation changed source termination evidence", state, err) + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + if kind != "missing" { + if err = durableJSON(path, r); err != nil { + t.Fatal(err) + } + } + before, readErr := os.ReadFile(path) + wantErr := sandbox.ErrOwnership + if kind == "missing" { + wantErr = os.ErrNotExist + if !errors.Is(readErr, os.ErrNotExist) { + t.Fatal("expected missing journal", readErr) + } + } else if readErr != nil { + t.Fatal(readErr) + } + if err = b.closeExpiredRestore(q); !errors.Is(err, wantErr) { + t.Fatalf("unsafe closure result: got %v, want %v", err, wantErr) + } + after, readErr := os.ReadFile(path) + if kind == "missing" { + if !errors.Is(readErr, os.ErrNotExist) { + t.Fatal("missing journal was created", readErr) + } + } else if readErr != nil || string(after) != string(before) { + t.Fatal("rejected closure changed journal", readErr) } }) } } + +func TestFreshAdmissionClosesOnlyAfterRequestDeadline(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + q := wire.ResumeRequest{OperationID: "66666666-6666-4666-8666-666666666666", Target: wire.Compute{Name: "target", Generation: 1}} + if _, err = b.admitRestore(q); err != nil { + t.Fatal(err) + } + if err = b.closeExpiredRestore(q); err != nil { + t.Fatal("live request was closed", err) + } + var r restoreAdmission + path := filepath.Join(b.storeDirectory(), "r-"+q.OperationID+".json") + if err = readPrivateJSON(path, &r); err != nil || r.Closed { + t.Fatal(r, err) + } + b.q.Deadline = time.Now().Add(-time.Second) + if err = b.closeExpiredRestore(q); !errors.Is(err, context.DeadlineExceeded) { + t.Fatal(err) + } + if err = readPrivateJSON(path, &r); err != nil || !r.Closed { + t.Fatal("expired dispatch not fenced", r, err) + } +} + +func TestCaptureAdmissionClosesNeverDispatchedOperationDurably(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + b := backend{q: request} + source := request.Compute + source.ID = "local:exact" + q := wire.SuspendRequest{OperationID: "66666666-6666-4666-8666-666666666666", Source: source, ObserveOnly: true} + if closed, err := b.admitSuspend(q); err != nil || !closed { + t.Fatal(closed, err) + } + guard.Close() + guard, err = allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + q.ObserveOnly = false + if closed, err := b.admitSuspend(q); err != nil || !closed { + t.Fatal("late fresh capture crossed durable closure", closed, err) + } + q.OperationID = "77777777-7777-4777-8777-777777777777" + if closed, err := b.admitSuspend(q); err != nil || closed { + t.Fatal(closed, err) + } + // A crash after open admission remains unknown even without a visible archive. + q.ObserveOnly = true + if closed, err := b.admitSuspend(q); err != nil || closed { + t.Fatal("unknown capture manufactured no-effect proof", closed, err) + } + q.ObserveOnly = false + if _, err := b.admitSuspend(q); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("admitted capture replayed", err) + } +} + +func TestCaptureAdmissionBoundAndSourceGenerationFence(t *testing.T) { + request := initialInfoRequest(t) + guard, err := allocationLock(request) + if err != nil { + t.Fatal(err) + } + defer guard.Close() + b := backend{q: request} + source := request.Compute + source.ID = "local:exact" + q := wire.SuspendRequest{Source: source, ObserveOnly: true} + for i := range 64 { + q.OperationID = fmt.Sprintf("%08x-6666-4666-8666-666666666666", i) + if closed, err := b.admitSuspend(q); err != nil || !closed { + t.Fatal(i, closed, err) + } + } + q.OperationID = "ffffffff-6666-4666-8666-666666666666" + if _, err := b.admitSuspend(q); !errors.Is(err, wire.ErrUnconfirmed) { + t.Fatal("capture journal exceeded its bound", err) + } + old := q + q.Source.Generation++ + q.Source.Name = wire.Name(request.Config, request.Reference, q.Source.Generation) + q.Source.ID = "local:next" + if closed, err := b.admitSuspend(q); err != nil || !closed { + t.Fatal("verified next source could not advance journal", closed, err) + } + old.ObserveOnly = false + if _, err := b.admitSuspend(old); !errors.Is(err, sandbox.ErrOwnership) { + t.Fatal("late older source admitted", err) + } + var journal captureAdmission + if err := readPrivateJSON(filepath.Join(b.storeDirectory(), "capture-admission.json"), &journal); err != nil || len(journal.Operations) != 1 || journal.Source.Generation != q.Source.Generation { + t.Fatal("generation journal not bounded", journal, err) + } +}