From 847bb38aab982434e382906f4348d5dd34c4956f Mon Sep 17 00:00:00 2001 From: "Mars.P" Date: Mon, 28 Sep 2026 10:04:08 +0800 Subject: [PATCH] Sever an allowlist run this worker cannot filter An allowlist run was planned Filtered wherever ip and nft were installed, which is all FilteredEgressNetns.IsSupported checks. The shipped worker is uid 1654 with no capabilities and ships both, so every allowlist run there was admitted, spent, and then aborted at `ip netns add` ("Filtered-egress netns setup failed"). A root worker whose /proc/sys is read-only fared worse: procps' `sysctl -w net.ipv4.ip_forward=1` warns and exits 0 there, so with forwarding at 0 the setup succeeded and the run launched as filtered with its allowed hosts unreachable. Where bubblewrap confines, the runner's egress derivation, and the admission that predicts it, now key on FilteredEgressNetns.CanFilter, renamed from CanSeal: the tools, the throwaway-namespace probe, and net.ipv4.ip_forward reading 1 or passing access(W_OK), which changes nothing. Where it does not hold, an allowlist fails closed to no egress, the contract SandboxEgressPolicy has always stated: bubblewrap severs the run, its record says NetworkSevered, and a brokered run still reaches its broker through the relay. The boot posture line reports CanFilter and FilterUnavailableReason, and the Dockerfile says what an allowlist needs and what a worker without it does. Where bubblewrap does not confine, as docker-compose.yml ships the worker, nothing would enforce that severance: the run would share the worker's network. There the binaries alone still plan it Filtered, so its setup filters it or aborts the launch, exactly as before. The launch and the admission read that one host fact (LocalProcessRunner.FiltersAllowlist). A worker that can filter still filters. Tearing a namespace down by name and the legacy re-bind still key on the tools alone, so runs in flight keep what they have. The non-root sandbox lane gains an allowlist arm that is admitted, launched with no namespace, recorded severed, and reaches its broker while the allowlisted IP stays unreachable; keyed back on the tools it aborts at setup again. A new unconfined lane (uid 1654, with CODESPACE_BWRAP_PATH naming no binary) pins that the same run there is aborted at setup, durable or not, and that the admission predicts its namespace; keyed on CanFilter everywhere, it launches on the worker's network. The root lane asserts its allowlist run is still filtered, and binds a file reading 0 read-only to show that the whole filter probe, not only its forwarding step, names forwarding root may not turn on. The non-root fixture also asserts its wall is the namespace, not forwarding, so a lane that can build one cannot pass as the shipped posture. --- .github/workflows/sandbox-isolation.yml | 67 ++++++-- backend/Dockerfile.worker | 26 +-- .../Sandbox/Isolation/CapabilityProbe.cs | 10 +- .../Sandbox/Isolation/FilteredEgressNetns.cs | 99 +++++++++--- .../Runners/LocalProcessRunner.Durable.cs | 54 +++++-- .../Sandbox/Runners/LocalProcessRunner.cs | 11 +- .../Jobs/RecurringJobWorkerSmokeE2ETests.cs | 2 +- .../AgentRunExecutorCredentialBrokerTests.cs | 8 +- .../DurableLaunchEgressE2ETests.cs | 94 +++++++++-- .../FilteredEgressNetnsE2ETests.cs | 61 ++++++- .../ModelCredentialBrokerNetnsE2ETests.cs | 2 +- .../CodeSpace.SandboxTests/NonRootWorker.cs | 3 +- .../NonRootWorkerE2ETests.cs | 12 +- .../SealedEgressE2ETests.cs | 2 +- .../UnconfinedWorkerE2ETests.cs | 107 ++++++++++++ .../Workflows/AllowlistFilterProbeTests.cs | 153 ++++++++++++++++++ .../Workflows/RootlessWorkerPostureTests.cs | 4 +- .../Workflows/SealedEgressAdmissionTests.cs | 32 ++++ 18 files changed, 658 insertions(+), 89 deletions(-) create mode 100644 backend/tests/CodeSpace.SandboxTests/UnconfinedWorkerE2ETests.cs create mode 100644 backend/tests/CodeSpace.UnitTests/Workflows/AllowlistFilterProbeTests.cs diff --git a/.github/workflows/sandbox-isolation.yml b/.github/workflows/sandbox-isolation.yml index 2597ea34b..06d7b3486 100644 --- a/.github/workflows/sandbox-isolation.yml +++ b/.github/workflows/sandbox-isolation.yml @@ -224,8 +224,8 @@ jobs: executed=$(grep -oE 'executed="[0-9]+"' "$trx" | head -1 | grep -oE '[0-9]+') passed=$(grep -oE 'passed="[0-9]+"' "$trx" | head -1 | grep -oE '[0-9]+') echo "executed=${executed:-0} passed=${passed:-0}" - if [ "${executed:-0}" -lt 82 ]; then - echo "::error::Expected >=82 sandbox isolation tests to run (bwrap/prlimit confinement + cap-drop + cgroup-namespace re-root + egress-allowlist filter + cgroup resource cap + durable-launch cgroup wiring + argv/envp per-string kernel ceiling + a prompt past it riding stdin + a read-only workspace mount + a network-off run reaching its broker through the relay and nothing else + an allowlist run relayed to its broker + a read-only reviewer reading its diff with the real CLIs + a bwrap probe that runs the launch argv + the MCP helper bound file by file behind a read-only socket dir + a CLI reaching its broker socket through the relay + a severed child reaching its broker over the lease socket across a worker restart + a pre-relay namespaced run re-bound at its gateway and torn down with its seal + an allowlist run's veth guarded both ways, a flow the worker opened before the run included), but only ${executed:-0} did — the Category=Sandbox filter matched too few (trait regression?). If a case was deliberately removed, lower this number in the same PR." + if [ "${executed:-0}" -lt 83 ]; then + echo "::error::Expected >=83 sandbox isolation tests to run (bwrap/prlimit confinement + cap-drop + cgroup-namespace re-root + egress-allowlist filter + cgroup resource cap + durable-launch cgroup wiring + argv/envp per-string kernel ceiling + a prompt past it riding stdin + a read-only workspace mount + a network-off run reaching its broker through the relay and nothing else + an allowlist run relayed to its broker + a read-only reviewer reading its diff with the real CLIs + a bwrap probe that runs the launch argv + the MCP helper bound file by file behind a read-only socket dir + a CLI reaching its broker socket through the relay + a severed child reaching its broker over the lease socket across a worker restart + a pre-relay namespaced run re-bound at its gateway and torn down with its seal + an allowlist run's veth guarded both ways, a flow the worker opened before the run included + forwarding a root worker may not write named before an allowlist is planned), but only ${executed:-0} did — the Category=Sandbox filter matched too few (trait regression?). If a case was deliberately removed, lower this number in the same PR." exit 1 fi @@ -265,14 +265,15 @@ jobs: assert len(cases) == rows and all(r.get('outcome') == 'Passed' for r in cases), f'{method}: all {rows} case(s) must pass' print('All 6 sealed-egress arms ran and passed.') - # The allowlist plan's host-routing arms, the guard on its veth, and the relayed allowlist launch return early - # without ip and nft; require their markers. The IPv6 arm prints its marker only where the veth has a link-local. - for marker in ('[filtered-egress-e2e] ran restart-reissue', '[filtered-egress-e2e] ran policy-route-discard-dst', '[filtered-egress-e2e] ran worker-shut', '[filtered-egress-e2e] ran ipv6-link-local', '[filtered-egress-e2e] ran shadowed-peer-refused', '[filtered-egress-e2e] ran established-flow-refused', '[filtered-egress-e2e] ran pmtu-upload', '[filtered-egress-e2e] ran stale-guard-replaced', '[durable-egress-e2e] ran allowlist-relay'): + # The allowlist plan's host-routing arms, the guard on its veth, the relayed allowlist launch, and the forwarding + # probe's read-only arm return early without ip and nft, or off root; require their markers. The IPv6 arm prints + # its marker only where the veth has a link-local. + for marker in ('[filtered-egress-e2e] ran restart-reissue', '[filtered-egress-e2e] ran policy-route-discard-dst', '[filtered-egress-e2e] ran worker-shut', '[filtered-egress-e2e] ran ipv6-link-local', '[filtered-egress-e2e] ran shadowed-peer-refused', '[filtered-egress-e2e] ran established-flow-refused', '[filtered-egress-e2e] ran pmtu-upload', '[filtered-egress-e2e] ran stale-guard-replaced', '[filtered-egress-e2e] ran forwarding-read-only', '[durable-egress-e2e] ran allowlist-relay'): assert marker in text, f'"{marker}" is missing — this lane is root with ip and nft, so the allowlist plan must run' - for method in ('FilteredEgressNetnsE2ETests.A_30_still_held_by_a_run_that_outlived_its_worker_is_not_handed_to_the_next_run', 'FilteredEgressNetnsE2ETests.A_host_whose_policy_rule_discards_the_run_s_replies_fails_the_setup_and_leaks_nothing', 'FilteredEgressNetnsE2ETests.An_allowlist_run_reaches_neither_the_worker_s_gateway_nor_its_address_while_dns_on_the_worker_and_the_allowlist_still_answer', 'FilteredEgressNetnsE2ETests.An_allowlist_run_has_no_path_to_the_worker_over_the_veth_s_ipv6_link_local_either', 'FilteredEgressNetnsE2ETests.A_peer_the_worker_reaches_at_the_run_s_address_is_refused_and_the_sandbox_receives_nothing', 'FilteredEgressNetnsE2ETests.A_flow_the_worker_opened_to_the_shadowed_peer_before_the_run_is_neither_handed_to_the_sandbox_nor_answered_from_it', 'FilteredEgressNetnsE2ETests.An_upload_across_a_narrower_uplink_completes_because_the_worker_s_frag_needed_reaches_the_run', 'FilteredEgressNetnsE2ETests.A_guard_an_earlier_teardown_left_behind_is_replaced_not_added_to', 'DurableLaunchEgressE2ETests.An_allowlist_run_reaches_its_broker_through_the_relay_and_its_allowlist_still_holds'): + for method in ('FilteredEgressNetnsE2ETests.A_30_still_held_by_a_run_that_outlived_its_worker_is_not_handed_to_the_next_run', 'FilteredEgressNetnsE2ETests.A_host_whose_policy_rule_discards_the_run_s_replies_fails_the_setup_and_leaks_nothing', 'FilteredEgressNetnsE2ETests.An_allowlist_run_reaches_neither_the_worker_s_gateway_nor_its_address_while_dns_on_the_worker_and_the_allowlist_still_answer', 'FilteredEgressNetnsE2ETests.An_allowlist_run_has_no_path_to_the_worker_over_the_veth_s_ipv6_link_local_either', 'FilteredEgressNetnsE2ETests.A_peer_the_worker_reaches_at_the_run_s_address_is_refused_and_the_sandbox_receives_nothing', 'FilteredEgressNetnsE2ETests.A_flow_the_worker_opened_to_the_shadowed_peer_before_the_run_is_neither_handed_to_the_sandbox_nor_answered_from_it', 'FilteredEgressNetnsE2ETests.An_upload_across_a_narrower_uplink_completes_because_the_worker_s_frag_needed_reaches_the_run', 'FilteredEgressNetnsE2ETests.A_guard_an_earlier_teardown_left_behind_is_replaced_not_added_to', 'FilteredEgressNetnsE2ETests.Forwarding_a_root_worker_may_not_write_is_named_before_an_allowlist_is_planned_on_it', 'DurableLaunchEgressE2ETests.An_allowlist_run_reaches_its_broker_through_the_relay_and_its_allowlist_still_holds'): cases = [r for r in results if method in r.get('testName', '')] assert len(cases) == 1 and cases[0].get('outcome') == 'Passed', f'{method}: must pass' - print('All 9 allowlist-plan arms ran and passed.') + print('All 10 allowlist-plan arms ran and passed.') # The broker's socket, relay-revoke and legacy-gateway arms return early without bwrap or ip/nft; require each # marker, the legacy survivor's teardown of the seal it carried included. @@ -327,9 +328,10 @@ jobs: - name: Test the non-root worker posture (uid 1654, no capabilities) # The shipped worker runs as uid 1654 with no capabilities and may not build a network namespace of its own, - # and every step above runs as root. Category=SandboxNonRoot runs the relay's arms again as that uid; each one - # first asserts geteuid() != 0, that bubblewrap confines, and that no namespace can be built, so the lane cannot - # pass as root or on a host that could fall back to a veth namespace. RequireConfinement stays set. The runner's + # and every step above runs as root. Category=SandboxNonRoot runs the relay's arms again as that uid, and an + # allowlist run it cannot filter, which is severed and still relayed; each one first asserts geteuid() != 0, + # that bubblewrap confines, and that no namespace can be built, so the lane cannot pass as root or on a host + # that could fall back to a veth namespace. RequireConfinement stays set. The runner's # kernel restricts unprivileged user namespaces through AppArmor where it has that switch; lifting it is the # same node setting operators give the worker. The root arms above leave their short-path socket roots behind # owner-only, which uid 1654 could not enter, so those go first. @@ -366,13 +368,53 @@ jobs: root = ET.parse(path).getroot() counters = root.find('.//{*}Counters') executed, passed = int(counters.get('executed')), int(counters.get('passed')) - assert executed >= 9 and passed == executed, f'expected all 9 non-root arms to run and pass, got executed={executed} passed={passed}' + assert executed >= 10 and passed == executed, f'expected all 10 non-root arms to run and pass, got executed={executed} passed={passed}' text = open(path, encoding='utf-8').read() - for marker in ('[non-root-e2e] ran admission uid=1654', '[sealed-egress-e2e] ran non-root durable uid=1654', '[sealed-egress-e2e] ran non-root non-durable uid=1654', '[sealed-egress-e2e] ran non-root restart uid=1654', '[sealed-egress-e2e] ran non-root relay-refused uid=1654', '[sealed-egress-e2e] ran non-root relay-ipv6', '[broker-socket-e2e] ran non-root socket-channel uid=1654', '[review-diff-e2e] ran non-root network-off-relayed claude-code uid=1654', '[review-diff-e2e] ran non-root network-off-relayed codex-cli uid=1654'): + for marker in ('[non-root-e2e] ran admission uid=1654', '[sealed-egress-e2e] ran non-root durable uid=1654', '[sealed-egress-e2e] ran non-root non-durable uid=1654', '[sealed-egress-e2e] ran non-root restart uid=1654', '[sealed-egress-e2e] ran non-root relay-refused uid=1654', '[sealed-egress-e2e] ran non-root relay-ipv6', '[broker-socket-e2e] ran non-root socket-channel uid=1654', '[review-diff-e2e] ran non-root network-off-relayed claude-code uid=1654', '[review-diff-e2e] ran non-root network-off-relayed codex-cli uid=1654', '[durable-egress-e2e] ran non-root allowlist-severed uid=1654'): assert marker in text, f'non-root arm marker "{marker}" is missing — the arm returned early or did not run as the worker uid' print(f'All {executed} non-root arms ran as uid 1654 and passed.') PY + - name: Test the unconfined worker posture (uid 1654, nothing confining) + # docker-compose.yml ships the worker image with none of the grants bubblewrap needs: uid 1654 with ip and nft + # installed, nothing confining, RequireConfinement off. A confining worker severs an allowlist it cannot filter; + # here nothing would enforce that, so Category=SandboxUnconfined pins that such a run is still planned into its + # namespace and aborted at the setup, never launched on the worker's network. It runs as that uid with + # CODESPACE_BWRAP_PATH naming no binary and RequireConfinement unset — the one lane that must not confine — + # and each arm first asserts exactly that posture. Reuses the home, packages and build output the non-root step + # made readable to that uid. + shell: bash + run: | + set -euo pipefail + results=backend/TestResults/unconfined + mkdir -p "$results" + chown 1654:1654 "$results" + packages=$(dotnet nuget locals global-packages --list | sed -E 's/^global-packages: *//; s:/+$::') + test -d "$packages" + setpriv --reuid 1654 --regid 1654 --clear-groups --inh-caps=-all --bounding-set=-all --no-new-privs \ + env -u Sandbox__RequireConfinement HOME=/tmp/nonroot-home DOTNET_CLI_HOME=/tmp/nonroot-home NUGET_PACKAGES="$packages" CODESPACE_BWRAP_PATH=/nonexistent/bwrap \ + dotnet test backend/tests/CodeSpace.SandboxTests/CodeSpace.SandboxTests.csproj \ + --no-build \ + --filter "Category=SandboxUnconfined" \ + --logger "console;verbosity=detailed" \ + --logger "trx;LogFileName=sandbox-unconfined.trx" \ + --results-directory "$results" + + - name: Assert the unconfined arms actually ran as uid 1654 + run: | + python3 - <<'PY' + import xml.etree.ElementTree as ET + path = 'backend/TestResults/unconfined/sandbox-unconfined.trx' + root = ET.parse(path).getroot() + counters = root.find('.//{*}Counters') + executed, passed = int(counters.get('executed')), int(counters.get('passed')) + assert executed >= 3 and passed == executed, f'expected all 3 unconfined arms to run and pass, got executed={executed} passed={passed}' + text = open(path, encoding='utf-8').read() + for marker in ('[unconfined-e2e] ran allowlist-never-unfiltered durable uid=1654', '[unconfined-e2e] ran allowlist-never-unfiltered one-shot uid=1654', '[unconfined-e2e] ran admission uid=1654'): + assert marker in text, f'unconfined arm marker "{marker}" is missing — the arm returned early or did not run as the worker uid' + print(f'All {executed} unconfined arms ran as uid 1654 and passed.') + PY + - name: Upload test results if: always() uses: actions/upload-artifact@v4 @@ -381,4 +423,5 @@ jobs: path: | backend/TestResults/*.trx backend/TestResults/nonroot/*.trx + backend/TestResults/unconfined/*.trx if-no-files-found: ignore diff --git a/backend/Dockerfile.worker b/backend/Dockerfile.worker index 60ef85d7c..ae7afd0af 100644 --- a/backend/Dockerfile.worker +++ b/backend/Dockerfile.worker @@ -23,8 +23,8 @@ # can still repoint CODESPACE_CODEX_CLI_PATH / CODESPACE_CLAUDE_CODE_PATH at a different binary. # # CONFINEMENT POSTURE. The worker logs one "Sandbox posture:" line at boot — whether bubblewrap confines and why not, -# whether the codespace-mcp helper is present, whether the namespace probe (FilteredEgressNetns.CanSeal) holds and why -# not. Read it before the first run; what each tier needs from the deployment is below. +# whether the codespace-mcp helper is present, whether the allowlist filter probe (FilteredEgressNetns.CanFilter) holds +# and why not. Read it before the first run; what each tier needs from the deployment is below. # # BUBBLEWRAP (every tier) needs no root and no capability: the non-root user below confines through UNPRIVILEGED user # namespaces, which a container runtime's defaults deny. Grant them, without --privileged or --cap-add: @@ -74,14 +74,20 @@ # (sandbox_sealed_egress_unavailable): so does one whose CODESPACE_MCP_PROXY_PATH names a build from before the relay # or a self-contained publish. An allowlist run reaches its broker through the same relay. # -# EGRESS FILTERING (an allowlist run) needs root + CAP_NET_ADMIN + CAP_SYS_ADMIN + a writable net.ipv4.ip_forward on -# top of these packages: FilteredEgressPlan runs `ip netns add` (CAP_SYS_ADMIN: it unshares a network namespace and -# bind-mounts it), builds a veth and an nftables ruleset, and `sysctl -w net.ipv4.ip_forward=1`. With `ip` or `nft` -# missing, FilteredEgressNetns.IsSupported is false and SandboxEgressPolicy.Derive turns the allowlist into Denied (NO -# network at all), never into Full. With them installed but the privilege missing — this image as shipped, since the -# user below is non-root — that probe still passes, so the run is planned Filtered and its launch aborts when setup is -# refused ("Filtered-egress netns setup failed"): it is never launched unfiltered, but it is not quietly Denied either. -# Grant the privilege on any deployment that relies on allowlisted egress. +# EGRESS FILTERING (an allowlist run) needs root + CAP_NET_ADMIN + CAP_SYS_ADMIN, and net.ipv4.ip_forward already 1 or +# writable, on top of these packages: FilteredEgressPlan runs `ip netns add` (CAP_SYS_ADMIN: it unshares a network +# namespace and bind-mounts it), builds a veth and an nftables ruleset, and `sysctl -w net.ipv4.ip_forward=1`, which +# only warns where /proc/sys is read-only or the write is denied, so forwarding that reads 0 there stays off. Where +# bubblewrap confines, the worker proves all of it before it plans a run Filtered (FilteredEgressNetns.CanFilter: the +# binaries, one throwaway namespace, and that sysctl), and the boot line names the step that failed. Where any of it is +# missing — this image as shipped with the grants above, since the user below is non-root — +# SandboxEgressPolicy.Derive turns the allowlist into Denied (NO network at all), never into Full: bubblewrap severs +# the run, its record says NetworkSevered, and a brokered one still reaches its model through the relay above. Where +# bubblewrap does not confine (none of the grants above, as docker-compose.yml ships it), nothing would enforce +# Denied, so the run is still planned Filtered on the binaries alone: its setup filters it, or is refused and aborts +# the launch ("Filtered-egress netns setup failed") — it is never launched on the worker's network, and +# Sandbox__RequireConfinement refuses such a host outright. Grant the privilege on any deployment that relies on +# allowlisted egress. # # global.json FLOORS the SDK feature band (a floor, not a full pin, against the floating sdk:10.0 tag). Multi-arch # base-image digest pinning is a deferred follow-up — it needs the manifest-LIST digest + a Renovate bump (a diff --git a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Isolation/CapabilityProbe.cs b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Isolation/CapabilityProbe.cs index 957f53bba..990377f9f 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Isolation/CapabilityProbe.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Isolation/CapabilityProbe.cs @@ -2,15 +2,15 @@ namespace CodeSpace.Core.Services.Agents.Sandbox.Isolation; /// /// Keeps the answer to a question this process can only settle by trying — are ip and nft runnable -/// (), can it build a sealed namespace (). +/// (), can it filter an allowlist run (). /// A proof holds for the process. A failure may be transient — a fork that failed once, a slow first mount of /// /run/netns, rtnl held by a burst of teardowns — so it stands for one retry interval and is then probed /// again, with its reason kept for whoever reports what could not be done. /// -/// The FIRST probe is waited for by every caller: a fresh worker that may well seal must not refuse its first -/// launches while it finds out. A RE-probe is made by one caller outside the lock while the others take the standing -/// failure — which they would have got anyway — instead of queueing behind a probe that can take tens of seconds on a -/// host that keeps failing slowly. Timed on a monotonic clock, so a wall-clock step cannot hold a failure forever or +/// The FIRST probe is waited for by every caller: a fresh worker that may well filter must not sever its first +/// allowlist runs while it finds out. A RE-probe is made by one caller outside the lock while the others take the +/// standing failure — which they would have got anyway — instead of queueing behind a probe that can take tens of +/// seconds on a host that keeps failing slowly. Timed on a monotonic clock, so a wall-clock step cannot hold a failure forever or /// re-probe on every call. /// internal sealed class CapabilityProbe(Func probe, Func monotonicNow, TimeSpan retryInterval) diff --git a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Isolation/FilteredEgressNetns.cs b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Isolation/FilteredEgressNetns.cs index bd42be5da..419a98c21 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Isolation/FilteredEgressNetns.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Isolation/FilteredEgressNetns.cs @@ -1,4 +1,5 @@ using System.Diagnostics; +using System.Runtime.InteropServices; using System.Text; namespace CodeSpace.Core.Services.Agents.Sandbox.Isolation; @@ -7,36 +8,98 @@ namespace CodeSpace.Core.Services.Agents.Sandbox.Isolation; /// The privileged executor of a (B3.2 enforcement) — sets up a per-run filtered /// network namespace, runs a command INSIDE it (so its only egress is the nftables allowlist), and tears the /// namespace down. Needs ip + nft + CAP_NET_ADMIN/root, so it runs for real only in the -/// privileged sandbox-isolation CI job; gates it everywhere else. Teardown is BEST-EFFORT +/// privileged sandbox-isolation CI job; gates it everywhere else bubblewrap confines, and the +/// binaries alone where nothing does (LocalProcessRunner.FiltersAllowlist). Teardown is BEST-EFFORT /// and ALWAYS runs (even on a setup failure mid-way), so a failed run never leaks a netns / veth / nft table. /// public static class FilteredEgressNetns { - /// How long a failed tools or seal probe stands before it is tried again (see ): caching a transient failure for the process lifetime would misreport this worker's posture until it restarted. - internal static readonly TimeSpan SealProbeRetryInterval = TimeSpan.FromMinutes(1); + /// How long a failed tools or filter probe stands before it is tried again (see ): caching a transient failure for the process lifetime would misreport this worker's posture until it restarted. + internal static readonly TimeSpan ProbeRetryInterval = TimeSpan.FromMinutes(1); + + /// The sysctl the plan's setup turns on (sysctl -w net.ipv4.ip_forward=1, ), as the file the filter probe reads. + internal const string IpForwardPath = "/proc/sys/net/ipv4/ip_forward"; + + private const string ToolsMissing = "ip or nft is not installed"; private static readonly long ProcessStart = System.Diagnostics.Stopwatch.GetTimestamp(); - private static readonly CapabilityProbe Tools = new(() => ProbeSupported() ? null : "ip or nft is not installed", () => System.Diagnostics.Stopwatch.GetElapsedTime(ProcessStart), SealProbeRetryInterval); + private static readonly CapabilityProbe Tools = new(() => ProbeSupported() ? null : ToolsMissing, () => System.Diagnostics.Stopwatch.GetElapsedTime(ProcessStart), ProbeRetryInterval); - private static readonly CapabilityProbe Seal = new(ProbeSeal, () => System.Diagnostics.Stopwatch.GetElapsedTime(ProcessStart), SealProbeRetryInterval); + private static readonly CapabilityProbe Filter = new(() => FilterProbe(IpForwardPath), () => System.Diagnostics.Stopwatch.GetElapsedTime(ProcessStart), ProbeRetryInterval); - /// True when ip + nft are present (the binaries the plan drives). Actual privilege to create a netns is exercised at run time — a setup failure fails closed. A failed probe is retried like the seal probe (): a fork that failed once at boot must not disable every allowlist, and the legacy re-bind's wide bind, for the process lifetime. + /// True when ip + nft are present (the binaries the plan drives). Whether this process may USE them to filter a run is , which a confining host's launch keys on; this alone gates what needs only the binaries — tearing a namespace down by name, the legacy re-bind's wide bind — and the allowlist plan of a host where bubblewrap does not confine, whose setup filters the run or aborts its launch (LocalProcessRunner.FiltersAllowlist). A failed probe is retried like the filter probe (): a fork that failed once at boot must not disable any of them for the process lifetime. public static bool IsSupported => Tools.Holds; /// - /// True when this process has PROVED it can build a namespace: the binaries are present AND one throwaway - /// namespace was created and deleted, and nftables answered. The binaries alone are not enough — an image can ship - /// them to a worker that runs without the privilege to use them. The boot posture line reports it; no launch keys - /// on it, since a namespaced run reaches its broker through a relay that needs no namespace of the worker's own. - /// A proof is kept for the process; a failure is kept for - /// and then probed again, and says why in . + /// True when this process has PROVED it can filter an allowlist run: the binaries are present, one throwaway + /// namespace was created and deleted and nftables answered, and forwarding will be on once the setup has asked for + /// it (). The binaries alone are not enough — an image can ship them to a + /// worker that runs without the privilege to use them, where an allowlist run planned Filtered would abort at its + /// setup after its spend was admitted. So where bubblewrap confines, the launch's egress derivation keys on this + /// (LocalProcessRunner.FiltersAllowlist): where it does not hold, an allowlist FAILS CLOSED to no egress + /// (), which bubblewrap enforces, and a brokered run still reaches its + /// broker through the relay. Where nothing confines, nothing would enforce that, so the binaries alone still plan + /// the run Filtered there. A proof is kept for the process; a failure is kept for + /// and then probed again, and says why in . /// - public static bool CanSeal => Seal.Holds; + public static bool CanFilter => Filter.Holds; - /// Why the last seal probe failed — the failed step and its output — or null when none has failed since the last proof. Read by the boot posture line, so the cause is not lost with the probe. - public static string? SealUnavailableReason => Seal.UnavailableReason; + /// Why the last filter probe failed — the failed step and its output — or null when none has failed since the last proof. Read by the boot posture line, so the cause is not lost with the probe. + public static string? FilterUnavailableReason => Filter.UnavailableReason; + + /// + /// The filter probe runs, with the forwarding file it asks about passed in + /// ( there), so the root lane can ask the whole composition about a file root may not + /// write, not only its parts. Its namespace step builds and deletes this process's one probe namespace for real, so + /// it is asked beside only once that has settled. + /// + internal static string? FilterProbe(string forwardingPath) => FilterProblem(IsSupported, ProbeNamespace, () => ForwardingProblem(forwardingPath)); + + /// + /// The first thing that stops this process filtering an allowlist run, in the order its setup meets them — the + /// binaries, a namespace, forwarding — or null when nothing does. A step is asked only once the one before it + /// holds, so a host without ip is never asked to run it. Pure over the three questions, so every row of the + /// truth table is testable on a host that can answer none of them. + /// + internal static string? FilterProblem(bool toolsPresent, Func namespaceProblem, Func forwardingProblem) + { + if (!toolsPresent) return ToolsMissing; + + return namespaceProblem() ?? forwardingProblem(); + } + + /// + /// Why forwarding would still be off after the setup's sysctl -w net.ipv4.ip_forward=1, or null when it + /// will be on: the file already reads 1, or this process may write it. Asked with a read and access(2), + /// which change nothing. The setup's own write cannot be the question: procps' sysctl -w warns and exits 0 + /// on a read-only /proc/sys and on a denied write alike, so a setup that ran it would succeed with nothing + /// forwarded — an allowlist whose allowed hosts are unreachable, recorded as filtered. + /// + internal static string? ForwardingProblem(string path) => ForwardingProblem(path, ReadForwarding(path), MayWrite(path)); + + /// over what the file read (null when it could not be read) and whether this process may write it. + internal static string? ForwardingProblem(string path, string? value, bool writable) + { + if (value == "1" || writable) return null; + + return value is null ? $"{path} could not be read, and this process may not write it" : $"{path} reads {value}, and this process may not write it"; + } + + private static string? ReadForwarding(string path) + { + try { return File.ReadAllText(path).Trim(); } + catch (Exception exception) when (exception is IOException or UnauthorizedAccessException) { return null; } + } + + private static bool MayWrite(string path) => !OperatingSystem.IsWindows() && Access(path, WriteOk) == 0; + + /// W_OK from unistd.h, the same value on Linux and macOS. + private const int WriteOk = 2; + + [DllImport("libc", EntryPoint = "access", SetLastError = true)] + private static extern int Access(string path, int mode); /// The outcome of running a command inside the filtered netns: the command's exit code + its combined output, plus whether the netns setup itself succeeded. public sealed record Outcome @@ -200,11 +263,9 @@ public static async Task RunAsync(string runId, IReadOnlyList a /// an add killed at its timeout may still have created it — so a probe can leave at most one namespace behind per /// worker, and the next probe from the same process removes it. /// - private static string? ProbeSeal() + private static string? ProbeNamespace() { - if (!IsSupported) return "ip or nft is not installed"; - - var probe = $"cs-seal-probe-{Environment.ProcessId}"; + var probe = $"cs-filter-probe-{Environment.ProcessId}"; try { diff --git a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.Durable.cs b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.Durable.cs index ee3392e91..8638d83e4 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.Durable.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.Durable.cs @@ -332,13 +332,15 @@ private static bool TryFileLength(string path, out long length) /// /// Derive this run's egress posture and, when it is an enforceable Filtered allowlist, set up the per-run netns and /// return the ip netns exec prefix the supervisor chain runs behind plus the teardown key. None/Full need no - /// netns (empty prefix, null key). Fail-closed: an allowlist requested on a runner that cannot enforce it degrades - /// to None (no netns) via , and a netns whose setup fails throws. A network-off run - /// never gets one: it reaches its broker, if it has one, through the relay (). + /// netns (empty prefix, null key). Fail-closed: an allowlist requested on a confining runner that cannot filter it + /// () degrades to None (no netns) via , where + /// bubblewrap severs it; on a runner where nothing confines, it is planned Filtered wherever the binaries are, and a + /// netns whose setup fails throws. A child with no netns reaches its broker, if it has one, through the relay + /// (). /// private static async Task<(IReadOnlyList ExecPrefix, string? Key)> SetupEgressNetnsAsync(SandboxSpec spec, string spoolKey, CancellationToken ct) { - var policy = SandboxEgressPolicy.Derive(spec.AllowNetwork, spec.EgressAllowlist, FilteredEgressNetns.IsSupported); + var policy = EgressPolicyFor(spec, HostFiltersAllowlist(BubblewrapSandbox.Available is not null)); if (policy.Mode != SandboxEgressMode.Filtered) return (Array.Empty(), null); @@ -362,14 +364,22 @@ private static bool TryFileLength(string path, out long length) /// at cannot run the relay there (). A spec with no /// broker port, or a child that shares the worker's network (an unconfined host, a network-granting run with no /// allowlist), is admitted untouched: nothing about its launch changes. The admission reads the same - /// the launch does, with the namespace predicted rather than built. + /// the launch does, with the namespace predicted rather than built — from the + /// same host fact the launch derives it from (), so a confining worker's + /// allowlist it cannot filter is admitted as the severed run it will be, never as a namespace its setup would then + /// refuse. /// - public void EnsureEgressAdmissible(SandboxSpec spec) => EnsureEgressAdmissible(spec, BubblewrapSandbox.Available is not null, McpProxyBinaryPath()); + public void EnsureEgressAdmissible(SandboxSpec spec) + { + var confines = BubblewrapSandbox.Available is not null; + + EnsureEgressAdmissible(spec, confines, HostFiltersAllowlist(confines), McpProxyBinaryPath()); + } - /// over whether this host and where it keeps the relay's helper, so a test can stand in for a confining host on any host. - internal static void EnsureEgressAdmissible(SandboxSpec spec, bool confines, string helperPath) + /// over whether this host , whether it plans an allowlist into a filtered namespace (, ), and where it keeps the relay's helper, so a test can stand in for any host on any host. + internal static void EnsureEgressAdmissible(SandboxSpec spec, bool confines, bool filtersAllowlist, string helperPath) { - if (RelayRefusal(spec, ChildNetworkIsPrivate(spec, WouldFilterEgress(spec), confines), helperPath) is { } cause) + if (RelayRefusal(spec, ChildNetworkIsPrivate(spec, WouldFilterEgress(spec, filtersAllowlist), confines), helperPath) is { } cause) throw new SealedEgressUnavailableException(cause); } @@ -403,8 +413,30 @@ internal static void EnsureEgressAdmissible(SandboxSpec spec, bool confines, str return ModelBrokerRelay.HelperRunsRelay(helperPath) ? null : $"{helperPath} did not answer as the relay: it predates it, or cannot start"; } - /// Whether this host launches inside a filtered-egress namespace — the prefix builds, predicted for the admission that runs before any of it exists. - private static bool WouldFilterEgress(SandboxSpec spec) => SandboxEgressPolicy.Derive(spec.AllowNetwork, spec.EgressAllowlist, FilteredEgressNetns.IsSupported).Mode == SandboxEgressMode.Filtered; + /// Whether a host that plans an allowlist into a filtered namespace () launches inside one — the prefix builds, predicted for the admission that runs before any of it exists. + private static bool WouldFilterEgress(SandboxSpec spec, bool filtersAllowlist) => EgressPolicyFor(spec, filtersAllowlist).Mode == SandboxEgressMode.Filtered; + + /// + /// The egress policy a host that plans an allowlist into a filtered namespace (, + /// ) gives : Filtered there, and otherwise None — severed, + /// never Full. The ONE derivation the launch and the admission read, so the admission cannot admit a run as the + /// namespace its setup would refuse. + /// + internal static SandboxEgressPolicy EgressPolicyFor(SandboxSpec spec, bool filtersAllowlist) => SandboxEgressPolicy.Derive(spec.AllowNetwork, spec.EgressAllowlist, filtersAllowlist); + + /// + /// Whether a host plans an allowlist run into a filtered namespace. Where bubblewrap , only + /// once every step of the setup is proven (, ): + /// bubblewrap severs the run it cannot filter, which still reaches its broker through the relay, where a setup it + /// then refused would abort the run after its spend was admitted. Where nothing confines, wherever the binaries are + /// (, ), as it always was: nothing there + /// would enforce None, so its setup filters the run or aborts its launch, and it is never launched on the worker's + /// network. + /// + internal static bool FiltersAllowlist(bool confines, bool canFilter, bool toolsPresent) => confines ? canFilter : toolsPresent; + + /// over this host's probes — the ONE host fact the launch and the admission read. The filter probe is asked only where bubblewrap confines, the one posture it decides. + private static bool HostFiltersAllowlist(bool confines) => FiltersAllowlist(confines, confines && FilteredEgressNetns.CanFilter, FilteredEgressNetns.IsSupported); /// /// Create this run's cgroup-v2 resource-cap leaf (B4) when a memory/cpu cap is requested AND the operator delegated diff --git a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs index 0cd1778e5..5c5b7e7bc 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs @@ -97,16 +97,17 @@ private sealed class AgentStalledException : Exception { } /// A BOOT diagnostic the worker host logs once, so the confinement posture runs will get is readable before the /// first run rather than reconstructed from a refused or unconfined one: whether bubblewrap confines and why not, /// whether the codespace-mcp helper that runs inside every sandbox is at , - /// and whether this process may build a network namespace () and why not. - /// These are the probes a launch reads, cached exactly as a launch caches them (bubblewrap's for the process, a - /// failed namespace probe for ), so this line is their first - /// caller, not a second opinion. Never throws: each probe already turns its own failure into a reason. + /// and whether this process can filter an allowlist run () and why not — + /// where it cannot and bubblewrap confines, such a run is severed instead. These are the probes a launch reads, + /// cached exactly as a launch caches them (bubblewrap's for the process, a failed filter probe for + /// ), so this line is their first caller, not a second opinion. + /// Never throws: each probe already turns its own failure into a reason. /// public static void LogSandboxPosture(ILogger logger) => LogSandboxPosture(logger, McpProxyBinaryPath()); /// with the helper path passed in, so a test pins both answers without mutating the process-wide override. internal static void LogSandboxPosture(ILogger logger, string helperPath) => - logger.LogInformation("Sandbox posture: bubblewrap confines {BubblewrapConfines} (unavailable reason: {BubblewrapUnavailableReason}); codespace-mcp helper present {McpProxyPresent} at {McpProxyPath}; namespace probe CanSeal {CanSeal} (unavailable reason: {SealUnavailableReason})", BubblewrapSandbox.Available is not null, BubblewrapSandbox.UnavailableReason ?? "none", File.Exists(helperPath), helperPath, FilteredEgressNetns.CanSeal, FilteredEgressNetns.SealUnavailableReason ?? "none"); + logger.LogInformation("Sandbox posture: bubblewrap confines {BubblewrapConfines} (unavailable reason: {BubblewrapUnavailableReason}); codespace-mcp helper present {McpProxyPresent} at {McpProxyPath}; allowlist filter probe CanFilter {CanFilter} (unavailable reason: {FilterUnavailableReason})", BubblewrapSandbox.Available is not null, BubblewrapSandbox.UnavailableReason ?? "none", File.Exists(helperPath), helperPath, FilteredEgressNetns.CanFilter, FilteredEgressNetns.FilterUnavailableReason ?? "none"); public async Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken) { diff --git a/backend/tests/CodeSpace.E2ETests/Jobs/RecurringJobWorkerSmokeE2ETests.cs b/backend/tests/CodeSpace.E2ETests/Jobs/RecurringJobWorkerSmokeE2ETests.cs index c63aff590..3c6c44a52 100644 --- a/backend/tests/CodeSpace.E2ETests/Jobs/RecurringJobWorkerSmokeE2ETests.cs +++ b/backend/tests/CodeSpace.E2ETests/Jobs/RecurringJobWorkerSmokeE2ETests.cs @@ -116,7 +116,7 @@ public async Task The_worker_host_logs_its_sandbox_posture_once_at_boot() + "no longer calls LocalProcessRunner.LogSandboxPosture (or calls it after an early return); more than one means it is called twice."); line.Level.ShouldBe(LogEventLevel.Information); - new[] { "BubblewrapConfines", "BubblewrapUnavailableReason", "McpProxyPresent", "CanSeal", "SealUnavailableReason" }.Except(line.Properties.Keys).ShouldBeEmpty( + new[] { "BubblewrapConfines", "BubblewrapUnavailableReason", "McpProxyPresent", "CanFilter", "FilterUnavailableReason" }.Except(line.Properties.Keys).ShouldBeEmpty( customMessage: "the posture line must carry each probe a launch reads, by the property names operators filter on"); } diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunExecutorCredentialBrokerTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunExecutorCredentialBrokerTests.cs index 737a1499f..886562637 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunExecutorCredentialBrokerTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/AgentRunExecutorCredentialBrokerTests.cs @@ -1608,9 +1608,9 @@ private async Task CreateTaskRunInWorkflowAsync(Guid teamId, Guid workflow /// /// A durable runner on a host that CONFINES, whatever host the test runs on: its admission is the local runner's - /// own (LocalProcessRunner.EnsureEgressAdmissible) with bubblewrap taken as present and the relay's helper at - /// . It builds no namespace, so its handles carry no namespace key — the non-root - /// worker's posture. Records what it was asked and what it launched. + /// own (LocalProcessRunner.EnsureEgressAdmissible) with bubblewrap taken as present, no allowlist it can + /// filter, and the relay's helper at . It builds no namespace, so its handles carry no + /// namespace key — the non-root worker's posture. Records what it was asked and what it launched. /// private sealed class ConfiningRunner(string helperPath) : ISandboxRunner, ISandboxDurableRunner, ISandboxEgressAdmission { @@ -1624,7 +1624,7 @@ public void EnsureEgressAdmissible(SandboxSpec spec) { Asked.Add(spec); - LocalProcessRunner.EnsureEgressAdmissible(spec, confines: true, helperPath); + LocalProcessRunner.EnsureEgressAdmissible(spec, confines: true, filtersAllowlist: false, helperPath); } public Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken) => diff --git a/backend/tests/CodeSpace.SandboxTests/DurableLaunchEgressE2ETests.cs b/backend/tests/CodeSpace.SandboxTests/DurableLaunchEgressE2ETests.cs index fcdd83325..512d45550 100644 --- a/backend/tests/CodeSpace.SandboxTests/DurableLaunchEgressE2ETests.cs +++ b/backend/tests/CodeSpace.SandboxTests/DurableLaunchEgressE2ETests.cs @@ -21,7 +21,9 @@ namespace CodeSpace.SandboxTests; /// reachable from inside the launched run, (2) a NON-allowed IP is DROPPED, and (3) the netns is REAPED on the /// run's terminal path (no leak). Needs ip + nft + CAP_NET_ADMIN, so it runs for real ONLY in the privileged /// sandbox-isolation CI job; elsewhere is false and it degrade-skips. -/// Uses raw IPs over plain HTTP so the signal is purely the egress filter — not DNS, not TLS. +/// Uses raw IPs over plain HTTP so the signal is purely the egress filter — not DNS, not TLS. The arm for a worker +/// that has the binaries but cannot filter () is run by the +/// non-root lane, which is that posture. /// [Trait("Category", "Sandbox")] public sealed class DurableLaunchEgressE2ETests(ITestOutputHelper output) @@ -63,22 +65,13 @@ public async Task An_allowlist_run_reaches_its_broker_through_the_relay_and_its_ // worker's own address, which the guard on the run's veth drops. if (!FilteredEgressNetns.IsSupported || BubblewrapSandbox.Available is null) return; // the root lane, with ip, nft and bwrap, is authoritative + FilteredEgressNetns.CanFilter.ShouldBeTrue($"ip and nft are here and this is the root lane, so an allowlist must be filterable here ({FilteredEgressNetns.FilterUnavailableReason}); otherwise it is severed, which the non-root lane pins"); + using var workerListener = new TcpListener(IPAddress.Any, 0); workerListener.Start(); - var runId = Guid.NewGuid(); - var permissions = new AgentPermissions { Network = AgentNetworkAccess.On, Egress = AgentEgressPolicy.Allowlist }; - var socketPath = AgentRunExecutor.ModelBrokerSocketPathFor(permissions, runId).ShouldNotBeNull("the executor mints a socket for an allowlist run on Linux"); using var broker = LoopbackModelCredentialBroker.ForTest(new OkUpstream()); - var brokered = (await broker.OpenAsync(new() { RunId = runId, TeamId = Guid.NewGuid(), Epoch = 1, Upstream = new() { Provider = "Anthropic", ApiKey = "sk-allowlist-e2e" }, Ttl = TimeSpan.FromMinutes(5), SocketPath = socketPath }, CancellationToken.None)).ShouldNotBeNull(); - var spec = AgentRunExecutor.ApplyModelBrokerChannel(new SandboxSpec - { - Command = "/usr/bin/python3", Args = ["-c", AllowlistProbe], AllowNetwork = true, EgressAllowlist = [Allowed], TimeoutSeconds = 60, - Environment = new Dictionary { ["BROKER_URL"] = brokered.BaseUrl, ["RUN_TOKEN"] = brokered.RunToken, ["SOCK_PATH"] = socketPath, ["WORKER_IP"] = SealedEgressE2ETests.WorkerIpv4(), ["WORKER_PORT"] = ((IPEndPoint)workerListener.LocalEndpoint).Port.ToString(CultureInfo.InvariantCulture) }, - }, brokered); - - spec.ModelBrokerSocketPath.ShouldBe(socketPath, "fixture check: the executor's own hardening stamps an allowlist run with its lease's socket"); - + var (spec, socketPath) = await BrokeredAllowlistSpecAsync(broker, workerListener); var key = Guid.NewGuid().ToString("N"); var runner = new LocalProcessRunner(); var lines = new List(); @@ -87,7 +80,8 @@ public async Task An_allowlist_run_reaches_its_broker_through_the_relay_and_its_ { var handle = await runner.LaunchAsync(spec, key, CancellationToken.None); handle.EgressNetnsKey.ShouldBe(key, "an enforceable allowlist still launches the run inside its filtered netns"); - handle.Confinement.ShouldNotBeNull().EgressSealedToBroker.ShouldBeFalse("an allowlist run is filtered, not sealed: it has more than one destination"); + handle.Confinement.ShouldNotBeNull().NetworkSevered.ShouldBeFalse("a filtered allowlist run shares its namespace's network: filtered, not severed"); + handle.Confinement.EgressSealedToBroker.ShouldBeFalse("an allowlist run is filtered, not sealed: it has more than one destination"); var result = await runner.AttachAsync(handle, (frame, _) => { lines.Add(frame.Text); return Task.CompletedTask; }, CancellationToken.None); var probe = string.Join(' ', lines); @@ -110,6 +104,78 @@ public async Task An_allowlist_run_reaches_its_broker_through_the_relay_and_its_ (await NetnsExistsAsync(NamespaceOf(key))).ShouldBeFalse("the run's filtered netns is reaped on completion"); } + /// + /// The severed arm, for the non-root lane (): a confining worker with ip + /// and nft installed that cannot filter — the shipped image's non-root posture — launches an allowlist run + /// severed, never into a namespace its setup would refuse after the run was admitted to spend (where nothing + /// confines, pins the other answer). The REAL runner + /// admits it, launches it with no namespace of the worker's and records it severed, and the probe inside it still + /// reaches its broker through the relay while the allowlisted IP is as unreachable as any other. + /// + internal async Task IsSeveredAndStillReachesItsBrokerAsync(string lane) + { + FilteredEgressNetns.IsSupported.ShouldBeTrue("fixture check: ip and nft are installed here, so the binaries alone would plan this run Filtered — the posture whose setup aborted it after its spend was admitted"); + FilteredEgressNetns.CanFilter.ShouldBeFalse("this lane cannot filter an allowlist, which is the posture this arm is for"); + + using var workerListener = new TcpListener(IPAddress.Any, 0); + workerListener.Start(); + + using var broker = LoopbackModelCredentialBroker.ForTest(new OkUpstream()); + var (spec, socketPath) = await BrokeredAllowlistSpecAsync(broker, workerListener); + var key = Guid.NewGuid().ToString("N"); + var runner = new LocalProcessRunner(); + var lines = new List(); + + Should.NotThrow(() => runner.EnsureEgressAdmissible(spec), "a severed allowlist run whose lease has its socket and whose helper runs the relay is admitted"); + + try + { + // Keyed on the binaries alone, the launch plans this run Filtered and throws here: its namespace setup is refused. + var handle = await runner.LaunchAsync(spec, key, CancellationToken.None); + handle.EgressNetnsKey.ShouldBeNull("an allowlist this worker cannot filter gets no namespace: bubblewrap severs it instead"); + + var record = handle.Confinement.ShouldNotBeNull(); + record.Outcome.ShouldBe(SandboxConfinementOutcome.Confined); + record.NetworkSevered.ShouldBeTrue("an allowlist this worker cannot filter FAILS CLOSED to no egress, and the run's record says so"); + + var result = await runner.AttachAsync(handle, (frame, _) => { lines.Add(frame.Text); return Task.CompletedTask; }, CancellationToken.None); + var probe = string.Join(' ', lines); + + result.Status.ShouldBe(SandboxStatus.Success, $"the probe must run to its end; stderr: {result.Stderr}"); + probe.ShouldContain("broker=200", customMessage: $"the severed run still reaches its broker through the relay; check `ls -la {Path.GetDirectoryName(socketPath)}`; probe: {probe}"); + probe.ShouldNotContain("allowed=open", customMessage: $"severed: the allowlisted IP is as unreachable as any other host, never reached unfiltered; probe: {probe}"); + probe.ShouldNotContain("denied=open", customMessage: $"and a host outside the allowlist is unreachable too; probe: {probe}"); + ProbeValue(probe, "worker").ShouldNotBe("open", $"and so is the worker's listener at the worker's own address; probe: {probe}"); + SealedEgressE2ETests.AssertSocketDirectoryIsReadOnly(ProbeValue(probe, "sock_unlink"), ProbeValue(probe, "sock_plant"), socketPath); + + output.WriteLine($"[durable-egress-e2e] ran {lane} allowlist-severed uid={NonRootWorker.EffectiveUid()} {probe} filter-unavailable=({FilteredEgressNetns.FilterUnavailableReason})"); + } + finally + { + try { Directory.Delete(LocalProcessRunner.SpoolDirectoryFor(key), recursive: true); } catch { /* best-effort */ } + } + + (await NetnsExistsAsync(NamespaceOf(key))).ShouldBeFalse("and no namespace was left behind for it"); + } + + /// An allowlist run of the probe below whose model is brokered, with the socket the executor mints for it on Linux, hardened as the executor hardens it, and told the worker's own address and 's port. + private static async Task<(SandboxSpec Spec, string SocketPath)> BrokeredAllowlistSpecAsync(LoopbackModelCredentialBroker broker, TcpListener workerListener) + { + var runId = Guid.NewGuid(); + var permissions = new AgentPermissions { Network = AgentNetworkAccess.On, Egress = AgentEgressPolicy.Allowlist }; + var socketPath = AgentRunExecutor.ModelBrokerSocketPathFor(permissions, runId).ShouldNotBeNull("the executor mints a socket for an allowlist run on Linux"); + var brokered = (await broker.OpenAsync(new() { RunId = runId, TeamId = Guid.NewGuid(), Epoch = 1, Upstream = new() { Provider = "Anthropic", ApiKey = "sk-allowlist-e2e" }, Ttl = TimeSpan.FromMinutes(5), SocketPath = socketPath }, CancellationToken.None)).ShouldNotBeNull(); + var spec = AgentRunExecutor.ApplyModelBrokerChannel(new SandboxSpec + { + Command = "/usr/bin/python3", Args = ["-c", AllowlistProbe], AllowNetwork = true, EgressAllowlist = [Allowed], TimeoutSeconds = 60, + Environment = new Dictionary { ["BROKER_URL"] = brokered.BaseUrl, ["RUN_TOKEN"] = brokered.RunToken, ["SOCK_PATH"] = socketPath, ["WORKER_IP"] = SealedEgressE2ETests.WorkerIpv4(), ["WORKER_PORT"] = ((IPEndPoint)workerListener.LocalEndpoint).Port.ToString(CultureInfo.InvariantCulture) }, + }, brokered); + + spec.ModelBrokerSocketPath.ShouldBe(socketPath, "fixture check: the executor's own hardening stamps an allowlist run with its lease's socket"); + + return (spec, socketPath); + } + /// The broker through the relay, the two writes the socket's read-only directory must refuse, the allowlisted IP and a denied one, and the worker's listener at the namespace's default gateway and at the worker's own address, from inside the run. private const string AllowlistProbe = SealedEgressE2ETests.SocketDirectoryWrites + "\n" + """ import os, socket, urllib.request diff --git a/backend/tests/CodeSpace.SandboxTests/FilteredEgressNetnsE2ETests.cs b/backend/tests/CodeSpace.SandboxTests/FilteredEgressNetnsE2ETests.cs index 206bb32de..5eac69638 100644 --- a/backend/tests/CodeSpace.SandboxTests/FilteredEgressNetnsE2ETests.cs +++ b/backend/tests/CodeSpace.SandboxTests/FilteredEgressNetnsE2ETests.cs @@ -28,7 +28,7 @@ namespace CodeSpace.SandboxTests; /// address or the veth's IPv6 link-local) while DNS on the worker and the allowlist still answer; a peer the worker /// reaches at an address the run's /30 shadows is refused and the sandbox receives nothing, a flow the worker opened to /// that peer before the run included; an upload across a narrower uplink still completes; and a guard an earlier round -/// left behind is replaced, not added to. +/// left behind is replaced, not added to. So does the one about forwarding a root worker may not turn on. /// [Trait("Category", "Sandbox")] public sealed class FilteredEgressNetnsE2ETests(ITestOutputHelper output) @@ -412,6 +412,63 @@ public async Task A_guard_an_earlier_teardown_left_behind_is_replaced_not_added_ finally { await FilteredEgressNetns.TeardownAsync(runId, CancellationToken.None); } } + [Fact] + public async Task Forwarding_a_root_worker_may_not_write_is_named_before_an_allowlist_is_planned_on_it() + { + // The setup's `sysctl -w net.ipv4.ip_forward=1` warns and exits 0 on a read-only /proc/sys, so on a root worker + // whose forwarding reads 0 there the setup succeeds and the allowlist reaches nothing. For root only a mount can + // refuse the write — file modes never do — so this binds a file reading 0 read-only and asks the real access(2), + // as the filter probe does of /proc/sys. Needs mount, so root alone; and a lane that builds namespaces, where the + // forwarding step alone can decide the whole probe — which is asked too, once CanFilter has settled, since both + // build this process's one probe namespace. + if (!OperatingSystem.IsLinux() || NonRootWorker.EffectiveUid() != 0 || !BuildsNamespaces()) return; + + using var file = new ReadOnlyForwardingFile(); + await file.StageAsync(); + + var refused = $"{file.ReadOnlyPath} reads 0, and this process may not write it"; + + FilteredEgressNetns.ForwardingProblem(file.WritablePath).ShouldBeNull("control: the same bytes where root may write them — the setup would turn forwarding on"); + FilteredEgressNetns.ForwardingProblem(file.ReadOnlyPath).ShouldBe(refused, customMessage: $"a read-only mount refuses root's write, so forwarding would stay off; check `grep {file.ReadOnlyPath} /proc/mounts`"); + + FilteredEgressNetns.FilterProbe(file.WritablePath).ShouldBeNull("control: the filter probe CanFilter runs holds on this lane where forwarding may be turned on"); + FilteredEgressNetns.FilterProbe(file.ReadOnlyPath).ShouldBe(refused, customMessage: "the probe CanFilter runs asks the forwarding step too: with a namespace buildable, forwarding root may not turn on is what stops an allowlist being filtered"); + + output.WriteLine($"{RanMarker} forwarding-read-only"); + } + + /// A file reading 0 and the same file bound read-only at a second path, both under a GUID-named directory; the bind is unmounted and the directory removed on dispose (Rule 12.2/12.3). + private sealed class ReadOnlyForwardingFile : IDisposable + { + private readonly string _dir = Path.Combine(Path.GetTempPath(), "cs-forwarding-" + Guid.NewGuid().ToString("N")); + + private bool _mounted; + + public string WritablePath => Path.Combine(_dir, "writable"); + + public string ReadOnlyPath => Path.Combine(_dir, "read-only"); + + public async Task StageAsync() + { + Directory.CreateDirectory(_dir); + await File.WriteAllTextAsync(WritablePath, "0\n"); + await File.WriteAllTextAsync(ReadOnlyPath, ""); + + (await RunHostExitAsync(["mount", "--bind", WritablePath, ReadOnlyPath])).ShouldBe(0, "setup: root must be able to bind a file over another"); + _mounted = true; + + (await RunHostExitAsync(["mount", "-o", "remount,ro,bind", ReadOnlyPath])).ShouldBe(0, "setup: and to make that bind read-only"); + } + + public void Dispose() + { + if (_mounted) + try { using var umount = System.Diagnostics.Process.Start("umount", ReadOnlyPath); umount.WaitForExit(10_000); } catch { /* best-effort */ } + + try { Directory.Delete(_dir, recursive: true); } catch { /* best-effort */ } + } + } + /// From inside the namespace: the worker's listener at the gateway and at its own address, DNS to the worker over UDP and TCP, and the allowlisted IP. private static async Task> ProbeTheWorkerAsync(FilteredEgressNetns.SetupResult setup, string workerIp, TcpListener listener) { @@ -832,7 +889,7 @@ private static bool BuildsNamespaces() { if (!FilteredEgressNetns.IsSupported) return false; - FilteredEgressNetns.CanSeal.ShouldBeTrue("ip and nft are here, but this process could not build a throwaway namespace — an allowlist run could not be filtered on this host"); + FilteredEgressNetns.CanFilter.ShouldBeTrue($"ip and nft are here, but this process could not filter an allowlist run on this host ({FilteredEgressNetns.FilterUnavailableReason})"); return true; } diff --git a/backend/tests/CodeSpace.SandboxTests/ModelCredentialBrokerNetnsE2ETests.cs b/backend/tests/CodeSpace.SandboxTests/ModelCredentialBrokerNetnsE2ETests.cs index 49f33f828..a40767996 100644 --- a/backend/tests/CodeSpace.SandboxTests/ModelCredentialBrokerNetnsE2ETests.cs +++ b/backend/tests/CodeSpace.SandboxTests/ModelCredentialBrokerNetnsE2ETests.cs @@ -289,7 +289,7 @@ public async Task A_gateway_addressed_run_launched_before_the_relay_is_re_bound_ // and ends by tearing the namespace down. if (!FilteredEgressNetns.IsSupported) return; // the root lane, with ip + nft, is authoritative - FilteredEgressNetns.CanSeal.ShouldBeTrue("ip and nft are here, but this process could not build a namespace — the survivor this arm stands for could not exist either"); + FilteredEgressNetns.CanFilter.ShouldBeTrue($"ip and nft are here, but this process could not build an allowlist namespace ({FilteredEgressNetns.FilterUnavailableReason}) — the survivor this arm stands for could not exist either"); var runId = Guid.NewGuid(); var teamId = Guid.NewGuid(); diff --git a/backend/tests/CodeSpace.SandboxTests/NonRootWorker.cs b/backend/tests/CodeSpace.SandboxTests/NonRootWorker.cs index aa5ca13d0..295a2ce3f 100644 --- a/backend/tests/CodeSpace.SandboxTests/NonRootWorker.cs +++ b/backend/tests/CodeSpace.SandboxTests/NonRootWorker.cs @@ -27,7 +27,8 @@ public static bool Require() GetEuid().ShouldNotBe(0u, "this is the non-root lane: run it as the worker's uid with no capabilities (setpriv --reuid 1654 --regid 1654 --clear-groups --inh-caps=-all --bounding-set=-all --no-new-privs); as root it proves nothing about the shipped posture"); BubblewrapSandbox.Available.ShouldNotBeNull($"bubblewrap must confine as this uid ({BubblewrapSandbox.UnavailableReason}); check `sysctl kernel.apparmor_restrict_unprivileged_userns` and `bwrap --unshare-user --unshare-net true` as this user"); - FilteredEgressNetns.CanSeal.ShouldBeFalse("this uid must NOT be able to build a network namespace — that is the posture the relay exists for; a worker that can was given CAP_SYS_ADMIN"); + FilteredEgressNetns.CanFilter.ShouldBeFalse("this uid must NOT be able to build a network namespace — that is the posture the relay exists for; a worker that can was given CAP_SYS_ADMIN"); + FilteredEgressNetns.FilterUnavailableReason.ShouldNotBeNull().ShouldNotStartWith(FilteredEgressNetns.IpForwardPath, customMessage: $"and the wall must be the namespace, not forwarding: this process built one and was stopped only at {FilteredEgressNetns.IpForwardPath}, so something gave it the privilege to build it (CAP_SYS_ADMIN, or a setuid ip)"); return true; } diff --git a/backend/tests/CodeSpace.SandboxTests/NonRootWorkerE2ETests.cs b/backend/tests/CodeSpace.SandboxTests/NonRootWorkerE2ETests.cs index c34fd833a..670eecef1 100644 --- a/backend/tests/CodeSpace.SandboxTests/NonRootWorkerE2ETests.cs +++ b/backend/tests/CodeSpace.SandboxTests/NonRootWorkerE2ETests.cs @@ -14,7 +14,9 @@ namespace CodeSpace.SandboxTests; /// process may not build a network namespace of its own. Until the relay, a confining host that could not build one /// refused every network-off brokered run before it spent; here the REAL runner admits it, and the real chain carries /// it to its broker through the relay and the lease's socket. Each arm asserts the posture first -/// (), then runs the SAME arm the root lane runs, so both lanes pin one behaviour. +/// (), then runs the SAME arm the root lane runs, so both lanes pin one behaviour — +/// but for the allowlist arm: this worker cannot filter, so its allowlist run is severed where the root lane's is +/// filtered, and it still reaches its broker. /// /// Selected by its trait alone (--filter Category=SandboxNonRoot), never by the root lane's /// Category=Sandbox. Every arm that ran prints its class's marker with non-root and its uid, which the @@ -79,6 +81,14 @@ public async Task The_same_live_agent_reaches_the_next_worker_through_its_socket await arms.ReachesTheNextWorkerAfterARestartAsync(Lane); } + [Fact] + public async Task An_allowlist_run_this_worker_cannot_filter_is_severed_and_still_reaches_its_broker_through_the_relay() + { + if (!NonRootWorker.Require()) return; + + await new DurableLaunchEgressE2ETests(output).IsSeveredAndStillReachesItsBrokerAsync(Lane); + } + [Fact] public async Task A_severed_child_reaches_its_broker_through_the_lease_socket_and_the_same_child_reaches_the_next_worker() { diff --git a/backend/tests/CodeSpace.SandboxTests/SealedEgressE2ETests.cs b/backend/tests/CodeSpace.SandboxTests/SealedEgressE2ETests.cs index a0ee1cdf4..dab2b1a11 100644 --- a/backend/tests/CodeSpace.SandboxTests/SealedEgressE2ETests.cs +++ b/backend/tests/CodeSpace.SandboxTests/SealedEgressE2ETests.cs @@ -250,7 +250,7 @@ internal async Task IsRefusedAndWouldHaveReachedNothingAsync(string lane) ((CodeSpace.Messages.Failures.IFailure)noSocket).Code.ShouldBe(CodeSpace.Messages.Failures.FailureCodes.SandboxSealedEgressUnavailable); var missing = Path.Combine(NewDirectory("no-helper"), "codespace-mcp"); - Should.Throw(() => LocalProcessRunner.EnsureEgressAdmissible(spec with { ModelBrokerSocketPath = "/spool/k/broker/seg/s" }, BubblewrapSandbox.Available is not null, missing)) + Should.Throw(() => LocalProcessRunner.EnsureEgressAdmissible(spec with { ModelBrokerSocketPath = "/spool/k/broker/seg/s" }, BubblewrapSandbox.Available is not null, FilteredEgressNetns.CanFilter, missing)) .Cause.ShouldStartWith(SealedEgressUnavailableException.CauseRelayMissing); var (result, probe) = await RunAsync(spec); diff --git a/backend/tests/CodeSpace.SandboxTests/UnconfinedWorkerE2ETests.cs b/backend/tests/CodeSpace.SandboxTests/UnconfinedWorkerE2ETests.cs new file mode 100644 index 000000000..c4b3bff3d --- /dev/null +++ b/backend/tests/CodeSpace.SandboxTests/UnconfinedWorkerE2ETests.cs @@ -0,0 +1,107 @@ +using System.Diagnostics; +using CodeSpace.Core.Services.Agents.Sandbox.Exceptions; +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; +using CodeSpace.Core.Services.Agents.Sandbox.Runners; +using CodeSpace.Messages.Agents; +using Shouldly; +using Xunit.Abstractions; + +namespace CodeSpace.SandboxTests; + +/// +/// 🟢 Sandbox isolation E2E (high fidelity, Rule 12), the UNCONFINED lane: the worker image as docker-compose.yml ships +/// it — uid 1654 with no capabilities, ip and nft installed, nothing letting bubblewrap confine, and +/// Sandbox:RequireConfinement off. A confining worker that cannot filter an allowlist severs it; here nothing +/// would enforce that, so severing would mean the worker's network. The REAL runner plans the run into its namespace +/// on the binaries alone, as it always has, so the setup's refusal aborts the launch — durable or not — and the +/// admission predicts that same namespace. +/// +/// Selected by its trait alone (--filter Category=SandboxUnconfined); the lane runs it as uid 1654 with +/// CODESPACE_BWRAP_PATH naming no binary and Sandbox__RequireConfinement unset. Each arm asserts that +/// posture first () and prints , which the lane requires. +/// +[Trait("Category", Category)] +public sealed class UnconfinedWorkerE2ETests(ITestOutputHelper output) +{ + /// The trait value the unconfined lane filters on. + public const string Category = "SandboxUnconfined"; + + /// Printed by every arm that ran. + public const string RanMarker = "[unconfined-e2e] ran"; + + [Theory] + [InlineData(true)] + [InlineData(false)] + public async Task An_allowlist_run_this_worker_cannot_filter_is_never_launched_on_the_worker_s_network(bool durable) + { + if (!RequirePosture()) return; + + var key = Guid.NewGuid().ToString("N"); + var spec = new SandboxSpec { Command = "/bin/true", AllowNetwork = true, EgressAllowlist = ["1.1.1.1"], TimeoutSeconds = 30 }; + var runner = new LocalProcessRunner(); + + Task Launch() => durable ? runner.LaunchAsync(spec, key, CancellationToken.None) : runner.RunAsync(spec, CancellationToken.None); + + try + { + var refused = await Should.ThrowAsync(Launch, customMessage: "severed here, the run would have shared the worker's network: nothing on this host enforces a severance"); + + refused.Message.ShouldContain("Filtered-egress netns setup failed", customMessage: $"planned into its namespace on the binaries alone, the launch aborts where the setup is refused; check `ip netns add probe` as uid {NonRootWorker.EffectiveUid()}"); + } + finally + { + try { Directory.Delete(LocalProcessRunner.SpoolDirectoryFor(key), recursive: true); } catch { /* best-effort */ } + } + + (await NetnsExistsAsync(FilteredEgressPlan.NamespaceFor(key))).ShouldBeFalse("and no namespace was left behind for it"); + + output.WriteLine($"{RanMarker} allowlist-never-unfiltered {(durable ? "durable" : "one-shot")} uid={NonRootWorker.EffectiveUid()} filter-unavailable=({FilteredEgressNetns.FilterUnavailableReason})"); + } + + [Fact] + public void The_admission_predicts_the_namespace_the_launch_plans() + { + if (!RequirePosture()) return; + + // Brokered, with no socket: only a child with a network of its own needs one, and here the launch plans this run + // into a namespace — so the admission must refuse it, from the same host fact, before anything is spent. + var spec = new SandboxSpec { Command = "agent", AllowNetwork = true, EgressAllowlist = ["api.anthropic.com"], ModelBrokerPort = 43121 }; + + var refused = Should.Throw(() => new LocalProcessRunner().EnsureEgressAdmissible(spec), "the admission must read the host fact the launch reads, which plans this run into its namespace here"); + + refused.Cause.ShouldBe(SealedEgressUnavailableException.CauseBrokerSocketUnavailable); + + output.WriteLine($"{RanMarker} admission uid={NonRootWorker.EffectiveUid()}"); + } + + /// + /// True on Linux once the posture is proved; false on any other OS (Rule 12.1). On Linux a missing piece FAILS the + /// test: an arm that ran where bubblewrap confines, or as root, proves nothing about the posture it is named for. + /// + private static bool RequirePosture() + { + if (!OperatingSystem.IsLinux()) return false; + + NonRootWorker.EffectiveUid().ShouldNotBe(0u, "this is the unconfined lane: run it as the worker's uid with no capabilities (setpriv --reuid 1654 --regid 1654 --clear-groups --inh-caps=-all --bounding-set=-all --no-new-privs)"); + BubblewrapSandbox.Available.ShouldBeNull($"nothing may confine in this lane: point {BubblewrapSandbox.CommandEnvVar} at no binary, as a runtime that denies bubblewrap leaves the worker"); + BubblewrapSandbox.IsRequired.ShouldBeFalse("Sandbox:RequireConfinement refuses every run on an unconfined host; the lane leaves it unset, as docker-compose.yml ships the worker"); + FilteredEgressNetns.IsSupported.ShouldBeTrue("ip and nft must be installed, as the image ships them: the binaries that plan an allowlist run into its namespace here"); + FilteredEgressNetns.CanFilter.ShouldBeFalse("and this uid must not be able to filter one, the posture where a severed allowlist would have meant the worker's network"); + + return true; + } + + /// True when is still a live network namespace (parsing ip netns list). + private static async Task NetnsExistsAsync(string ns) + { + var psi = new ProcessStartInfo { FileName = "ip", UseShellExecute = false, RedirectStandardOutput = true, RedirectStandardError = true }; + psi.ArgumentList.Add("netns"); + psi.ArgumentList.Add("list"); + + using var p = Process.Start(psi)!; + var listing = await p.StandardOutput.ReadToEndAsync(); + await p.WaitForExitAsync(); + + return listing.Split('\n').Any(line => line.Trim().Split(' ').FirstOrDefault() == ns); + } +} diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AllowlistFilterProbeTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AllowlistFilterProbeTests.cs new file mode 100644 index 000000000..813d68958 --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Workflows/AllowlistFilterProbeTests.cs @@ -0,0 +1,153 @@ +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; +using CodeSpace.Core.Services.Agents.Sandbox.Runners; +using CodeSpace.Messages.Agents; +using Shouldly; + +namespace CodeSpace.UnitTests.Workflows; + +/// +/// Pins what "this worker can filter an allowlist run" means () and what the +/// launch derives from it (): where bubblewrap confines, every step of +/// the allowlist setup must be able to run — the binaries, a namespace, forwarding — or the run is severed and relayed, +/// never planned into a namespace its setup would refuse after the run was admitted to spend; where nothing confines, +/// the binaries alone plan it, as they always did. Pure over the answers, so every row is pinned on any host; the real +/// kernel is the sandbox lanes' (DurableLaunchEgressE2ETests: filtered as root, severed as the non-root worker; +/// UnconfinedWorkerE2ETests: never on the worker's network where nothing confines). +/// +[Trait("Category", "Unit")] +public sealed class AllowlistFilterProbeTests : IDisposable +{ + private const string ToolsMissing = "ip or nft is not installed"; + + private readonly string _dir = Directory.CreateTempSubdirectory("cs-filter-probe-").FullName; + + [Theory] + [InlineData(true, true, true, null)] // every step can run: filtered + [InlineData(true, true, false, "no forwarding")] // a namespace, but nothing would be forwarded out of it + [InlineData(true, false, true, "no namespace")] // the binaries without the privilege to use them: the shipped non-root worker + [InlineData(true, false, false, "no namespace")] // the first wall the setup meets is the one named + [InlineData(false, true, true, ToolsMissing)] // no ip or nft + [InlineData(false, true, false, ToolsMissing)] + [InlineData(false, false, true, ToolsMissing)] + [InlineData(false, false, false, ToolsMissing)] + public void Where_bubblewrap_confines_an_allowlist_is_filtered_only_where_every_step_of_its_setup_can_run_and_severed_everywhere_else(bool tools, bool buildsNamespace, bool forwards, string? wall) + { + var namespaceAsked = false; + + var problem = FilteredEgressNetns.FilterProblem(tools, () => { namespaceAsked = true; return buildsNamespace ? null : "no namespace"; }, () => forwards ? null : "no forwarding"); + + problem.ShouldBe(wall); + namespaceAsked.ShouldBe(tools, "a host without ip is never asked to run it"); + + var canFilter = problem is null; + var filters = LocalProcessRunner.FiltersAllowlist(confines: true, canFilter, tools); + var spec = new SandboxSpec { Command = "agent", AllowNetwork = true, EgressAllowlist = ["api.anthropic.com"], ModelBrokerPort = 43121, ModelBrokerSocketPath = "/spool/k/broker/seg/s" }; + + filters.ShouldBe(canFilter, "where bubblewrap confines, the filter probe alone decides"); + LocalProcessRunner.EgressPolicyFor(spec, filters).Mode.ShouldBe(canFilter ? SandboxEgressMode.Filtered : SandboxEgressMode.None, "an allowlist that cannot be filtered FAILS CLOSED to no egress, never to the worker's network"); + + // The setup builds a namespace, and so a prefix, for a Filtered run alone; what each posture then launches: + var prefix = filters ? new[] { "ip", "netns", "exec", "cs-egr-deadbeef" } : Array.Empty(); + var record = LocalProcessRunner.LaunchConfinement(spec, prefix, "/usr/bin/bwrap", unavailableReason: null); + + record.NetworkSevered.ShouldBe(!canFilter, "filtered in its namespace, or severed by bubblewrap — and the record says which"); + record.EgressSealedToBroker.ShouldBeFalse("sealed stays a network-off run's word: an allowlist run asked for network, and severing it is the fail-closed answer"); + LocalProcessRunner.RelaysModelBroker(spec, prefix.Length > 0, confines: true).ShouldBeTrue("filtered or severed, its network is its own, so it reaches its broker through the relay either way"); + } + + [Theory] + [InlineData(true, true, true, true)] // bubblewrap confines and every step can run: filtered + [InlineData(true, false, true, false)] // bubblewrap confines, the binaries without the privilege: severed and relayed — the shipped non-root worker + [InlineData(true, false, false, false)] // bubblewrap confines, no ip or nft: severed + [InlineData(false, true, true, true)] // nothing confines, every step can run: filtered, its namespace its only network + [InlineData(false, false, true, true)] // nothing confines and the setup would be refused: planned as it always was, so the setup aborts the launch — nothing here would enforce None + [InlineData(false, false, false, false)] // nothing confines, no ip or nft (macOS dev): as it always was + public void Where_nothing_confines_the_binaries_alone_plan_an_allowlist_so_it_is_never_launched_on_the_worker_s_network(bool confines, bool canFilter, bool toolsPresent, bool filtered) + { + var spec = new SandboxSpec { Command = "agent", AllowNetwork = true, EgressAllowlist = ["api.anthropic.com"] }; + + var filters = LocalProcessRunner.FiltersAllowlist(confines, canFilter, toolsPresent); + + filters.ShouldBe(filtered); + + // Planned Filtered, the launch runs in its namespace or aborts at its setup; otherwise bubblewrap severs it, or — + // with no ip or nft to plan it and nothing to confine it, as before — it shares the worker's network. + LocalProcessRunner.ChildNetworkIsPrivate(spec, filters, confines).ShouldBe(confines || toolsPresent, "a host with the binaries never launches an allowlist on the worker's network, confined or not"); + } + + [Theory] + [InlineData("1", false, true)] // already on: the setup's write, taken or ignored, changes nothing + [InlineData("1", true, true)] + [InlineData("0", true, true)] // off, and this process may turn it on: the setup does + [InlineData("0", false, false)] // off, and the setup's write would be ignored (a read-only /proc/sys): nothing forwarded + [InlineData(null, true, true)] // unreadable, but the write takes + [InlineData(null, false, false)] // neither + public void Forwarding_is_on_after_the_setup_where_it_already_reads_1_or_this_process_may_write_it(string? value, bool writable, bool forwards) + { + var problem = FilteredEgressNetns.ForwardingProblem(FilteredEgressNetns.IpForwardPath, value, writable); + + if (forwards) problem.ShouldBeNull(); + else problem.ShouldNotBeNull().ShouldStartWith(FilteredEgressNetns.IpForwardPath, customMessage: "the boot line names the file an operator has to make writable"); + } + + [Fact] + public void A_real_file_that_already_reads_1_needs_no_write_even_one_this_process_may_not_make() + { + if (OperatingSystem.IsWindows()) return; + + var on = Path.Combine(_dir, "on"); + File.WriteAllText(on, "1\n"); + File.SetUnixFileMode(on, UnixFileMode.UserRead); + + // Root writes whatever the mode says, so there this row would pass through the writable branch instead: the pure + // row ("1", false) pins the branch on every host, and a host that honours the mode pins it on a real file. + if (OpensForWrite(on)) return; + + FilteredEgressNetns.ForwardingProblem(on).ShouldBeNull("a file that already reads 1 needs no write"); + } + + [Fact] + public void The_forwarding_question_is_asked_of_the_real_file_and_changes_nothing() + { + if (OperatingSystem.IsWindows()) return; + + var off = Path.Combine(_dir, "off"); + File.WriteAllText(off, "0\n"); + + FilteredEgressNetns.ForwardingProblem(off).ShouldBeNull("a file this process may write is one the setup's sysctl turns on"); + File.ReadAllText(off).ShouldBe("0\n", "asking changes nothing: the setup, not the probe, turns forwarding on"); + + var absent = Path.Combine(_dir, "absent"); + + FilteredEgressNetns.ForwardingProblem(absent).ShouldBe($"{absent} could not be read, and this process may not write it"); + } + + [Fact] + public void The_forwarding_probe_reads_the_sysctl_the_setup_writes() + { + var plan = FilteredEgressPlan.Build("run-forwarding", ["1.1.1.1"], new EgressSubnetAllocator.Lease { Cidr = "10.5.7.16/30", HostIp = "10.5.7.17", NsIp = "10.5.7.18" }); + + var write = plan.SetupCommands.Single(argv => argv[0] == "sysctl"); + write.Take(2).ShouldBe(new[] { "sysctl", "-w" }); + + var setting = write[2].Split('='); + + FilteredEgressNetns.IpForwardPath.ShouldBe("/proc/sys/" + setting[0].Replace('.', '/'), "the probe asks about the key the setup writes"); + setting[1].ShouldBe("1", "and the value it writes is the one the probe takes as already on"); + } + + private static bool OpensForWrite(string path) + { + try + { + using var stream = new FileStream(path, FileMode.Open, FileAccess.Write); + return true; + } + catch (UnauthorizedAccessException) { return false; } + } + + public void Dispose() + { + try { Directory.Delete(_dir, recursive: true); } catch { /* best-effort */ } + } +} diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/RootlessWorkerPostureTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/RootlessWorkerPostureTests.cs index 923b2204e..6cc957793 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/RootlessWorkerPostureTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/RootlessWorkerPostureTests.cs @@ -105,8 +105,8 @@ public void The_boot_posture_line_names_each_probe_a_launch_reads(bool helperPre line.Properties["BubblewrapUnavailableReason"].ShouldBe(BubblewrapSandbox.UnavailableReason ?? "none"); line.Properties["McpProxyPresent"].ShouldBe(helperPresent); line.Properties["McpProxyPath"].ShouldBe(helperPath); - line.Properties["CanSeal"].ShouldBe(FilteredEgressNetns.CanSeal); - line.Properties["SealUnavailableReason"].ShouldBe(FilteredEgressNetns.SealUnavailableReason ?? "none"); + line.Properties["CanFilter"].ShouldBe(FilteredEgressNetns.CanFilter); + line.Properties["FilterUnavailableReason"].ShouldBe(FilteredEgressNetns.FilterUnavailableReason ?? "none"); } /// Allowed for every caller on every architecture: no argument filter, and no capability, architecture or kernel condition. diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/SealedEgressAdmissionTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/SealedEgressAdmissionTests.cs index 55a8156ca..7fad0e05d 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/SealedEgressAdmissionTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/SealedEgressAdmissionTests.cs @@ -108,6 +108,38 @@ public void The_admission_reads_the_same_private_network_rule_as_the_launch(bool LocalProcessRunner.ChildNetworkIsPrivate(spec, inNamespace, confines).ShouldBe(expected); } + [Theory] + [InlineData(true, false, Helper.Relay, null)] // a confining worker that cannot filter: severed, and relayed — admitted as the run it will launch + [InlineData(true, false, Helper.Missing, SealedEgressUnavailableException.CauseRelayMissing)] // severed, with no helper to relay it: refused before it spends + [InlineData(true, true, Helper.Relay, null)] // filtered in its namespace, and relayed + [InlineData(false, true, Helper.Missing, SealedEgressUnavailableException.CauseRelayMissing)] // in its namespace on a host with no bwrap, it still needs the relay + [InlineData(false, false, Helper.Missing, null)] // nothing confines it and no ip or nft plan it into a namespace: it shares the worker's network and calls loopback itself + public void An_allowlist_run_is_admitted_as_the_launch_this_worker_can_give_it(bool confines, bool filtersAllowlist, Helper helper, string? expected) + { + if (OperatingSystem.IsWindows()) return; + + var spec = new SandboxSpec { Command = "agent", AllowNetwork = true, EgressAllowlist = ["api.anthropic.com"], ModelBrokerPort = 43121, ModelBrokerSocketPath = "/spool/k/broker/seg/s" }; + + var thrown = Record.Exception(() => LocalProcessRunner.EnsureEgressAdmissible(spec, confines, filtersAllowlist, HelperAt(helper))); + + if (expected is null) thrown.ShouldBeNull(); + else thrown.ShouldBeOfType().Cause.ShouldStartWith(expected, customMessage: "the cause leads with the wall"); + } + + [Fact] + public void A_brokered_allowlist_spec_without_its_socket_is_refused_exactly_where_this_host_would_give_it_a_network_of_its_own() + { + // The allowlist twin of the network-off row below, over this host's own probes: a confining host severs or + // filters the run, and a host where nothing confines plans it into a namespace wherever ip and nft are, so + // either way nothing could carry it to its broker. Only an unconfined host without the binaries admits it. + var spec = new SandboxSpec { Command = "agent", AllowNetwork = true, EgressAllowlist = ["api.anthropic.com"], ModelBrokerPort = 43121 }; + + var thrown = Record.Exception(() => new LocalProcessRunner().EnsureEgressAdmissible(spec)); + + if (BubblewrapSandbox.Available is null && !FilteredEgressNetns.IsSupported) thrown.ShouldBeNull(); + else thrown.ShouldBeOfType().Cause.ShouldBe(SealedEgressUnavailableException.CauseBrokerSocketUnavailable); + } + [Fact] public void A_spec_with_no_broker_port_is_admitted_on_any_host() => // Nothing to reach: every unbrokered run launches exactly as it always did.