diff --git a/backend/Dockerfile.worker b/backend/Dockerfile.worker index 192756ab7..e4c19e2d3 100644 --- a/backend/Dockerfile.worker +++ b/backend/Dockerfile.worker @@ -22,26 +22,65 @@ # CodexHarness / ClaudeCodeHarness default to `codex` / `claude` on PATH, so no env wiring is needed; an operator # can still repoint CODESPACE_CODEX_CLI_PATH / CODESPACE_CLAUDE_CODE_PATH at a different binary. # -# EGRESS FILTERING needs more than these packages. FilteredEgressNetns probes `ip -Version` + `nft --version` and, -# when either is missing, reports IsSupported=false — at which point SandboxEgressPolicy.Derive turns a run that asked -# for an EGRESS ALLOWLIST into Denied (NO network at all), never into Full. That is fail-closed and safe, but it means -# a worker image without these packages silently makes the allowlist feature unusable rather than loud. Installing -# them is necessary, NOT sufficient: `ip netns add` and `sysctl -w net.ipv4.ip_forward=1` need root + CAP_NET_ADMIN, -# and the runtime user below is non-root — so on a pod without those the probe may pass while setup fails. Grant -# CAP_NET_ADMIN (and run the egress path privileged enough to write the sysctl) on any deployment that relies on -# allowlisted egress; a pod that does not is still correct, it just gets Denied instead of Filtered. +# CONFINEMENT POSTURE. The worker logs one "Sandbox posture:" line at boot — whether bubblewrap confines and why not, +# whether the codespace-mcp helper is present, whether the namespace probe (FilteredEgressNetns.CanSeal) holds and why +# not. Read it before the first run; what each tier needs from the deployment is below. # -# The same namespace machinery SEALS a network-off run whose model is brokered to that broker (no route, no NAT, no -# DNS, one gateway port). It is taken only where bubblewrap confines AND FilteredEgressNetns.CanSeal has proved, by -# building a throwaway namespace, that this process may: root + CAP_NET_ADMIN + CAP_SYS_ADMIN, no sysctl needed. A -# pod without them that DOES confine refuses such a run before it spends anything (sandbox_sealed_egress_unavailable), -# because severing it instead would cut it off from its broker and leave it reaching no model. +# BUBBLEWRAP (every tier) needs no root and no capability: the non-root user below confines through UNPRIVILEGED user +# namespaces, which a container runtime's defaults deny. Grant them, without --privileged or --cap-add: +# • Docker / Compose (Compose v2.15+, which first accepts systempaths) — the commented block in docker-compose.yml: +# --security-opt seccomp=backend/deploy/seccomp/codespace-worker.json moby's v20.10.20 default as it applies to +# a container with no capabilities, plus clone/clone3/unshare with +# namespace flags, setns, mount, umount2, pivot_root (the default +# reserves those for CAP_SYS_ADMIN; pivot_root it denies). It is in +# the OCI form with no capability or architecture conditions, so +# Docker, containerd and CRI-O apply the same filter, and a +# capability the container holds adds no call. Regenerate it with +# derive-codespace-worker.sh beside it, never by hand. +# --security-opt apparmor=unconfined the runtime's AppArmor profile denies mount +# --security-opt systempaths=unconfined bubblewrap mounts a fresh /proc, which the kernel refuses over a +# masked one ("Can't mount proc on /newroot/proc") +# • Kubernetes 1.33+ (user namespaces and ProcMountType on by default; node kernel 6.3+, containerd 2.0+ or CRI-O 1.25+): +# spec: +# hostUsers: false +# containers: +# - name: worker +# securityContext: +# runAsNonRoot: true +# runAsUser: 1654 +# allowPrivilegeEscalation: false +# capabilities: { drop: ["ALL"] } +# procMount: Unmasked +# seccompProfile: { type: Localhost, localhostProfile: codespace-worker.json } # in each node's kubelet seccomp dir +# appArmorProfile: { type: Unconfined } # on AppArmor nodes +# Older clusters: securityContext { privileged: true, runAsUser: 1654 }. That confines too, but it lifts the +# container's own seccomp, AppArmor and capability bounds, so prefer the form above. Pod Security baseline and +# restricted reject AppArmor Unconfined, and whether they admit procMount: Unmasked for a hostUsers: false pod +# depends on the cluster version; check yours before labelling the worker's namespace. +# • Nodes: user.max_user_namespaces above 0 (EKS Auto Mode sets 0, so nothing confines there); Ubuntu 23.10+ nodes +# also need kernel.apparmor_restrict_unprivileged_userns=0 or an AppArmor profile that lets bwrap create userns. +# Where user namespaces or mounts are denied the bubblewrap probe fails and runs are UNCONFINED, recorded as such on +# every run. A masked /proc is different: the probe mounts no /proc, so with every grant but systempaths=unconfined / +# procMount: Unmasked the boot line reads "bubblewrap confines True" while every launch fails with "Can't mount proc on +# /newroot/proc"; check that the first run's agent actually starts. The image does not arm +# Sandbox:RequireConfinement, because that would refuse every run on such a host; a deployment that grants the above +# arms it (Sandbox__RequireConfinement=true), so a lost grant refuses runs instead of unconfining them. # -# CONFINEMENT ARMING: bubblewrap is installed here so the capability is PRESENT, but the fail-closed guard -# Sandbox:RequireConfinement is left to the DEPLOYMENT to arm (k8s pod / compose) on a host that grants user -# namespaces — hardcoding it in the image would fail-close every run on an environment where in-container userns -# is not configured. The non-root user below runs bubblewrap via UNPRIVILEGED user namespaces — validate that the -# target host/pod permits unprivileged userns for this uid (else arm RequireConfinement only where it does). +# NETWORK-OFF runs (Confined and Standard, the default tier) are severed by bubblewrap (--unshare-net: loopback only). +# One whose model is brokered still has to reach its broker. Once the model-broker relay ships (`codespace-mcp relay`), +# it does so through a per-run Unix socket bound read-only into the sandbox, which needs no root and nothing beyond +# the grants above. Without the relay it is SEALED to the broker through a per-run namespace instead (no route, no NAT, +# no DNS, one gateway port), which needs root + CAP_NET_ADMIN + CAP_SYS_ADMIN (FilteredEgressNetns.CanSeal); a worker +# that confines but cannot seal refuses such a run before it spends anything (sandbox_sealed_egress_unavailable). +# +# EGRESS FILTERING (an allowlist run) needs root + CAP_NET_ADMIN + CAP_SYS_ADMIN + a writable net.ipv4.ip_forward on +# top of these packages: FilteredEgressPlan runs `ip netns add` (CAP_SYS_ADMIN: it unshares a network namespace and +# bind-mounts it), builds a veth and an nftables ruleset, and `sysctl -w net.ipv4.ip_forward=1`. With `ip` or `nft` +# missing, FilteredEgressNetns.IsSupported is false and SandboxEgressPolicy.Derive turns the allowlist into Denied (NO +# network at all), never into Full. With them installed but the privilege missing — this image as shipped, since the +# user below is non-root — that probe still passes, so the run is planned Filtered and its launch aborts when setup is +# refused ("Filtered-egress netns setup failed"): it is never launched unfiltered, but it is not quietly Denied either. +# Grant the privilege on any deployment that relies on allowlisted egress. # # global.json FLOORS the SDK feature band (a floor, not a full pin, against the floating sdk:10.0 tag). Multi-arch # base-image digest pinning is a deferred follow-up — it needs the manifest-LIST digest + a Renovate bump (a diff --git a/backend/deploy/seccomp/codespace-worker.json b/backend/deploy/seccomp/codespace-worker.json new file mode 100644 index 000000000..ab3a2c7d8 --- /dev/null +++ b/backend/deploy/seccomp/codespace-worker.json @@ -0,0 +1,484 @@ +{ + "defaultAction": "SCMP_ACT_ERRNO", + "architectures": [ + "SCMP_ARCH_X86_64", + "SCMP_ARCH_X86", + "SCMP_ARCH_X32", + "SCMP_ARCH_AARCH64", + "SCMP_ARCH_ARM" + ], + "syscalls": [ + { + "names": [ + "accept", + "accept4", + "access", + "adjtimex", + "alarm", + "bind", + "brk", + "capget", + "capset", + "chdir", + "chmod", + "chown", + "chown32", + "clock_adjtime", + "clock_adjtime64", + "clock_getres", + "clock_getres_time64", + "clock_gettime", + "clock_gettime64", + "clock_nanosleep", + "clock_nanosleep_time64", + "close", + "close_range", + "connect", + "copy_file_range", + "creat", + "dup", + "dup2", + "dup3", + "epoll_create", + "epoll_create1", + "epoll_ctl", + "epoll_ctl_old", + "epoll_pwait", + "epoll_pwait2", + "epoll_wait", + "epoll_wait_old", + "eventfd", + "eventfd2", + "execve", + "execveat", + "exit", + "exit_group", + "faccessat", + "faccessat2", + "fadvise64", + "fadvise64_64", + "fallocate", + "fanotify_mark", + "fchdir", + "fchmod", + "fchmodat", + "fchown", + "fchown32", + "fchownat", + "fcntl", + "fcntl64", + "fdatasync", + "fgetxattr", + "flistxattr", + "flock", + "fork", + "fremovexattr", + "fsetxattr", + "fstat", + "fstat64", + "fstatat64", + "fstatfs", + "fstatfs64", + "fsync", + "ftruncate", + "ftruncate64", + "futex", + "futex_time64", + "futex_waitv", + "futimesat", + "getcpu", + "getcwd", + "getdents", + "getdents64", + "getegid", + "getegid32", + "geteuid", + "geteuid32", + "getgid", + "getgid32", + "getgroups", + "getgroups32", + "getitimer", + "getpeername", + "getpgid", + "getpgrp", + "getpid", + "getppid", + "getpriority", + "getrandom", + "getresgid", + "getresgid32", + "getresuid", + "getresuid32", + "getrlimit", + "get_robust_list", + "getrusage", + "getsid", + "getsockname", + "getsockopt", + "get_thread_area", + "gettid", + "gettimeofday", + "getuid", + "getuid32", + "getxattr", + "inotify_add_watch", + "inotify_init", + "inotify_init1", + "inotify_rm_watch", + "io_cancel", + "ioctl", + "io_destroy", + "io_getevents", + "io_pgetevents", + "io_pgetevents_time64", + "ioprio_get", + "ioprio_set", + "io_setup", + "io_submit", + "io_uring_enter", + "io_uring_register", + "io_uring_setup", + "ipc", + "kill", + "landlock_add_rule", + "landlock_create_ruleset", + "landlock_restrict_self", + "lchown", + "lchown32", + "lgetxattr", + "link", + "linkat", + "listen", + "listxattr", + "llistxattr", + "_llseek", + "lremovexattr", + "lseek", + "lsetxattr", + "lstat", + "lstat64", + "madvise", + "membarrier", + "memfd_create", + "memfd_secret", + "mincore", + "mkdir", + "mkdirat", + "mknod", + "mknodat", + "mlock", + "mlock2", + "mlockall", + "mmap", + "mmap2", + "mprotect", + "mq_getsetattr", + "mq_notify", + "mq_open", + "mq_timedreceive", + "mq_timedreceive_time64", + "mq_timedsend", + "mq_timedsend_time64", + "mq_unlink", + "mremap", + "msgctl", + "msgget", + "msgrcv", + "msgsnd", + "msync", + "munlock", + "munlockall", + "munmap", + "nanosleep", + "newfstatat", + "_newselect", + "open", + "openat", + "openat2", + "pause", + "pidfd_open", + "pidfd_send_signal", + "pipe", + "pipe2", + "poll", + "ppoll", + "ppoll_time64", + "prctl", + "pread64", + "preadv", + "preadv2", + "prlimit64", + "process_mrelease", + "pselect6", + "pselect6_time64", + "pwrite64", + "pwritev", + "pwritev2", + "read", + "readahead", + "readlink", + "readlinkat", + "readv", + "recv", + "recvfrom", + "recvmmsg", + "recvmmsg_time64", + "recvmsg", + "remap_file_pages", + "removexattr", + "rename", + "renameat", + "renameat2", + "restart_syscall", + "rmdir", + "rseq", + "rt_sigaction", + "rt_sigpending", + "rt_sigprocmask", + "rt_sigqueueinfo", + "rt_sigreturn", + "rt_sigsuspend", + "rt_sigtimedwait", + "rt_sigtimedwait_time64", + "rt_tgsigqueueinfo", + "sched_getaffinity", + "sched_getattr", + "sched_getparam", + "sched_get_priority_max", + "sched_get_priority_min", + "sched_getscheduler", + "sched_rr_get_interval", + "sched_rr_get_interval_time64", + "sched_setaffinity", + "sched_setattr", + "sched_setparam", + "sched_setscheduler", + "sched_yield", + "seccomp", + "select", + "semctl", + "semget", + "semop", + "semtimedop", + "semtimedop_time64", + "send", + "sendfile", + "sendfile64", + "sendmmsg", + "sendmsg", + "sendto", + "setfsgid", + "setfsgid32", + "setfsuid", + "setfsuid32", + "setgid", + "setgid32", + "setgroups", + "setgroups32", + "setitimer", + "setpgid", + "setpriority", + "setregid", + "setregid32", + "setresgid", + "setresgid32", + "setresuid", + "setresuid32", + "setreuid", + "setreuid32", + "setrlimit", + "set_robust_list", + "setsid", + "setsockopt", + "set_thread_area", + "set_tid_address", + "setuid", + "setuid32", + "setxattr", + "shmat", + "shmctl", + "shmdt", + "shmget", + "shutdown", + "sigaltstack", + "signalfd", + "signalfd4", + "sigprocmask", + "sigreturn", + "socket", + "socketcall", + "socketpair", + "splice", + "stat", + "stat64", + "statfs", + "statfs64", + "statx", + "symlink", + "symlinkat", + "sync", + "sync_file_range", + "syncfs", + "sysinfo", + "tee", + "tgkill", + "time", + "timer_create", + "timer_delete", + "timer_getoverrun", + "timer_gettime", + "timer_gettime64", + "timer_settime", + "timer_settime64", + "timerfd_create", + "timerfd_gettime", + "timerfd_gettime64", + "timerfd_settime", + "timerfd_settime64", + "times", + "tkill", + "truncate", + "truncate64", + "ugetrlimit", + "umask", + "uname", + "unlink", + "unlinkat", + "utime", + "utimensat", + "utimensat_time64", + "utimes", + "vfork", + "vmsplice", + "wait4", + "waitid", + "waitpid", + "write", + "writev" + ], + "action": "SCMP_ACT_ALLOW" + }, + { + "names": [ + "clone", + "clone3", + "mount", + "pivot_root", + "setns", + "umount2", + "unshare" + ], + "action": "SCMP_ACT_ALLOW", + "comment": "CodeSpace worker: the calls bubblewrap makes to build its user-namespace sandbox, which moby reserves for CAP_SYS_ADMIN (pivot_root it denies). See backend/Dockerfile.worker." + }, + { + "names": [ + "ptrace" + ], + "action": "SCMP_ACT_ALLOW" + }, + { + "names": [ + "personality" + ], + "action": "SCMP_ACT_ALLOW", + "args": [ + { + "index": 0, + "value": 0, + "op": "SCMP_CMP_EQ" + } + ] + }, + { + "names": [ + "personality" + ], + "action": "SCMP_ACT_ALLOW", + "args": [ + { + "index": 0, + "value": 8, + "op": "SCMP_CMP_EQ" + } + ] + }, + { + "names": [ + "personality" + ], + "action": "SCMP_ACT_ALLOW", + "args": [ + { + "index": 0, + "value": 131072, + "op": "SCMP_CMP_EQ" + } + ] + }, + { + "names": [ + "personality" + ], + "action": "SCMP_ACT_ALLOW", + "args": [ + { + "index": 0, + "value": 131080, + "op": "SCMP_CMP_EQ" + } + ] + }, + { + "names": [ + "personality" + ], + "action": "SCMP_ACT_ALLOW", + "args": [ + { + "index": 0, + "value": 4294967295, + "op": "SCMP_CMP_EQ" + } + ] + }, + { + "names": [ + "sync_file_range2" + ], + "action": "SCMP_ACT_ALLOW" + }, + { + "names": [ + "arm_fadvise64_64", + "arm_sync_file_range", + "sync_file_range2", + "breakpoint", + "cacheflush", + "set_tls" + ], + "action": "SCMP_ACT_ALLOW" + }, + { + "names": [ + "arch_prctl" + ], + "action": "SCMP_ACT_ALLOW" + }, + { + "names": [ + "modify_ldt" + ], + "action": "SCMP_ACT_ALLOW" + }, + { + "names": [ + "s390_pci_mmio_read", + "s390_pci_mmio_write", + "s390_runtime_instr" + ], + "action": "SCMP_ACT_ALLOW" + } + ] +} diff --git a/backend/deploy/seccomp/derive-codespace-worker.sh b/backend/deploy/seccomp/derive-codespace-worker.sh new file mode 100755 index 000000000..2564d9b1b --- /dev/null +++ b/backend/deploy/seccomp/derive-codespace-worker.sh @@ -0,0 +1,53 @@ +#!/usr/bin/env bash +# Regenerate codespace-worker.json, the seccomp profile that lets the non-root, capability-less worker build +# bubblewrap's sandbox (see backend/Dockerfile.worker), from moby v20.10.20's default profile: +# +# curl -fsSLo /tmp/moby-default.json https://raw.githubusercontent.com/moby/moby/v20.10.20/profiles/seccomp/default.json +# backend/deploy/seccomp/derive-codespace-worker.sh /tmp/moby-default.json +# +# The output is moby's default RESOLVED for a container with no capabilities, in the OCI runtime-spec form. moby's +# file gates rules on capabilities, architectures and kernel versions (includes / excludes / archMap), and only Docker +# and CRI-O evaluate those. containerd decodes a Localhost profile straight into the OCI LinuxSeccomp struct and drops +# every key it lacks, so each capability-gated allow would become an unconditional one there. Resolving the conditions +# here makes Docker, containerd and CRI-O apply the same filter: +# • rules gated on a capability (includes.caps) are dropped: the worker holds none; +# • rules gated on an architecture or kernel (includes.arches, includes.minKernel) keep their allow unconditionally: +# runc skips a name the native architecture does not have, and the one kernel gate (ptrace, 4.8) predates every +# kernel that runs the worker; +# • archMap becomes an explicit little-endian architectures list, the x86_64 and aarch64 families moby's archMap +# names; runc refuses a filter that mixes in a big-endian one (EDOM); +# • moby's clone rules that mask out namespace flags and its clone3 ENOSYS rule are replaced by one unconditional +# allow of the seven calls bubblewrap needs. +# +# RootlessWorkerPostureTests pins the output's SHA-256; update that pin in the same diff as any change here. +set -euo pipefail + +MOBY_DEFAULT_SHA256=ce3585b856aa4a193671fd8280d0d0612a998bd3f10bdb2f8530dfa958c6197f + +src="${1:?usage: $0 }" +dst="$(cd "$(dirname "$0")" && pwd)/codespace-worker.json" + +actual=$({ sha256sum "$src" 2>/dev/null || shasum -a 256 "$src"; } | cut -d' ' -f1) +[ "$actual" = "$MOBY_DEFAULT_SHA256" ] || { echo "✗ $src is not moby v20.10.20's default profile (sha256 $actual)"; exit 1; } + +jq --tab ' + def bubblewrap: { + names: ["clone", "clone3", "mount", "pivot_root", "setns", "umount2", "unshare"], + action: "SCMP_ACT_ALLOW", + comment: "CodeSpace worker: the calls bubblewrap makes to build its user-namespace sandbox, which moby reserves for CAP_SYS_ADMIN (pivot_root it denies). See backend/Dockerfile.worker." + }; + def replaced_by_bubblewrap: (.names == ["clone"] and .action == "SCMP_ACT_ALLOW" and (.args | length) > 0) or (.names == ["clone3"] and .action == "SCMP_ACT_ERRNO"); + def needs_a_capability: (.includes.caps // []) | length > 0; + def unresolved: ((.excludes // {}) | length > 0) or ((.includes // {}) | keys - ["arches", "caps", "minKernel"] | length > 0); + def oci_rule: {names, action} + (if .errnoRet then {errnoRet} else {} end) + (if (.args // []) | length > 0 then {args} else {} end); + + [.syscalls[] | select((replaced_by_bubblewrap or needs_a_capability) | not)] as $kept + | if any($kept[]; unresolved) then error("a kept rule carries a condition this script does not resolve") else . end + | { + defaultAction, + architectures: [.archMap[] | select(.architecture == "SCMP_ARCH_X86_64" or .architecture == "SCMP_ARCH_AARCH64") | .architecture, .subArchitectures[]], + syscalls: ([$kept[0] | oci_rule, bubblewrap] + [$kept[1:][] | oci_rule]) + } +' "$src" > "$dst" + +echo "✓ wrote $dst (sha256 $({ sha256sum "$dst" 2>/dev/null || shasum -a 256 "$dst"; } | cut -d' ' -f1))" diff --git a/backend/src/CodeSpace.Api/Extensions/Hangfire/HangfireRegistrar.Worker.cs b/backend/src/CodeSpace.Api/Extensions/Hangfire/HangfireRegistrar.Worker.cs index 8c9c374a4..cf2d1a47c 100644 --- a/backend/src/CodeSpace.Api/Extensions/Hangfire/HangfireRegistrar.Worker.cs +++ b/backend/src/CodeSpace.Api/Extensions/Hangfire/HangfireRegistrar.Worker.cs @@ -2,6 +2,7 @@ using CodeSpace.Messages.Enums; using CodeSpace.Core.Jobs; using CodeSpace.Core.Services.Agents; +using CodeSpace.Core.Services.Agents.Sandbox.Runners; using CodeSpace.Core.Services.Jobs; using Hangfire; using Microsoft.Extensions.DependencyInjection; @@ -73,6 +74,10 @@ public override void ApplyHangfire(IApplicationBuilder app, IConfiguration confi // hours later. No-op + never-throws when the endpoint is off. AgentRunExecutor.LogMcpProxyReadiness(app.ApplicationServices.GetRequiredService().CreateLogger()); + // WORKER-ONLY for the same reason: the confinement posture (bubblewrap, the in-sandbox helper, the namespace + // probe) decides how every run here launches, so an operator reads it at boot instead of from the first run. + LocalProcessRunner.LogSandboxPosture(app.ApplicationServices.GetRequiredService().CreateLogger()); + // A non-processing pod must NOT own recurring-job scheduling/execution. ScanHangfireRecurringJobs(app); } diff --git a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs index 2810a4e1b..0cd1778e5 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Sandbox/Runners/LocalProcessRunner.cs @@ -1,6 +1,7 @@ using System.Diagnostics; using System.Text; using CodeSpace.Core.DependencyInjection; +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; using CodeSpace.Messages.Agents; using Microsoft.Extensions.Logging; using Microsoft.Extensions.Logging.Abstractions; @@ -92,6 +93,21 @@ private sealed class AgentStalledException : Exception { } public string Kind => LocalKind; + /// + /// A BOOT diagnostic the worker host logs once, so the confinement posture runs will get is readable before the + /// first run rather than reconstructed from a refused or unconfined one: whether bubblewrap confines and why not, + /// whether the codespace-mcp helper that runs inside every sandbox is at , + /// and whether this process may build a network namespace () and why not. + /// These are the probes a launch reads, cached exactly as a launch caches them (bubblewrap's for the process, a + /// failed namespace probe for ), so this line is their first + /// caller, not a second opinion. Never throws: each probe already turns its own failure into a reason. + /// + public static void LogSandboxPosture(ILogger logger) => LogSandboxPosture(logger, McpProxyBinaryPath()); + + /// with the helper path passed in, so a test pins both answers without mutating the process-wide override. + internal static void LogSandboxPosture(ILogger logger, string helperPath) => + logger.LogInformation("Sandbox posture: bubblewrap confines {BubblewrapConfines} (unavailable reason: {BubblewrapUnavailableReason}); codespace-mcp helper present {McpProxyPresent} at {McpProxyPath}; namespace probe CanSeal {CanSeal} (unavailable reason: {SealUnavailableReason})", BubblewrapSandbox.Available is not null, BubblewrapSandbox.UnavailableReason ?? "none", File.Exists(helperPath), helperPath, FilteredEgressNetns.CanSeal, FilteredEgressNetns.SealUnavailableReason ?? "none"); + public async Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken) { if (spec.CaptureBudget is not null) return await RunWithBoundedCaptureAsync(new BoundedCommandRequest(spec, null), cancellationToken).ConfigureAwait(false); diff --git a/backend/tests/CodeSpace.E2ETests/Infrastructure/RecurringJobWorkerHostFactory.cs b/backend/tests/CodeSpace.E2ETests/Infrastructure/RecurringJobWorkerHostFactory.cs index 9f83e42ea..3e5b39390 100644 --- a/backend/tests/CodeSpace.E2ETests/Infrastructure/RecurringJobWorkerHostFactory.cs +++ b/backend/tests/CodeSpace.E2ETests/Infrastructure/RecurringJobWorkerHostFactory.cs @@ -1,4 +1,5 @@ using CodeSpace.Core.Persistence.Db; +using CodeSpace.Core.Services.Agents.Sandbox.Runners; using CodeSpace.Core.Settings; using CodeSpace.Messages.Enums; using Microsoft.AspNetCore.Hosting; @@ -8,6 +9,7 @@ using Npgsql; using Serilog; using Serilog.Core; +using Serilog.Events; namespace CodeSpace.E2ETests.Infrastructure; @@ -142,11 +144,12 @@ protected override void ConfigureWebHost(IWebHostBuilder builder) /// Appended AFTER Program.CreateHostBuilder's own UseSerilog(), so this host's /// ILoggerFactory registration is the last one and wins — every ILogger<T> the pipeline /// resolves (including TransactionalBehavior's) writes to the sink. The test proves that wiring with a - /// canary line before it trusts the absence of anything. + /// canary line before it trusts the absence of anything. Warning and above, except + /// at Information, which is the level of the worker's boot "Sandbox posture:" line. /// protected override IHost CreateHost(IHostBuilder builder) { - builder.UseSerilog(new LoggerConfiguration().MinimumLevel.Warning().WriteTo.Sink(_logSink).CreateLogger(), dispose: true); + builder.UseSerilog(new LoggerConfiguration().MinimumLevel.Warning().MinimumLevel.Override(typeof(LocalProcessRunner).FullName!, LogEventLevel.Information).WriteTo.Sink(_logSink).CreateLogger(), dispose: true); return base.CreateHost(builder); } diff --git a/backend/tests/CodeSpace.E2ETests/Jobs/RecurringJobWorkerSmokeE2ETests.cs b/backend/tests/CodeSpace.E2ETests/Jobs/RecurringJobWorkerSmokeE2ETests.cs index 65e6c1533..c63aff590 100644 --- a/backend/tests/CodeSpace.E2ETests/Jobs/RecurringJobWorkerSmokeE2ETests.cs +++ b/backend/tests/CodeSpace.E2ETests/Jobs/RecurringJobWorkerSmokeE2ETests.cs @@ -40,6 +40,9 @@ namespace CodeSpace.E2ETests.Jobs; /// relaxes the production boot guards), and it binds the host's own Serilog logger so the log assertion can read /// what the pipeline wrote. It also sets one process-wide environment variable at init, like its sibling fixtures. /// +/// Because this is the one fixture that boots the Worker role, it also pins the worker's boot "Sandbox posture:" +/// line (). +/// /// Deliberately NOT covered: cron TIMING (whether a cadence is right), job payloads, and what any individual /// sweep does to rows. Cron-fired EXECUTIONS are covered — they land in the same database and are judged by /// . Per-sweep behaviour belongs with each sweep's own tests, which can @@ -92,6 +95,33 @@ public async Task Every_recurring_job_the_worker_registers_completes_one_tick() AssertNothingRolledBack(sink); } + /// + /// The worker's boot line naming its confinement posture: Dockerfile.worker and docker-compose.yml tell an + /// operator to read it before the first run, so a registrar refactor that stops emitting it must go red here, + /// where the real WorkerHangfireRegistrar.ApplyHangfire runs. What each value says is pinned by + /// RootlessWorkerPostureTests; this pins that the worker host says it, once. + /// + [Fact] + public async Task The_worker_host_logs_its_sandbox_posture_once_at_boot() + { + var sink = new CapturedLogLines(); + + await using var factory = new RecurringJobWorkerHostFactory(sink); + await factory.InitializeAsync(); + + _ = factory.Services; // builds and starts the host, which runs ApplyHangfire + + var line = sink.Events().Where(IsSandboxPostureLine).ToList().ShouldHaveSingleItem( + customMessage: "the worker host must log exactly one 'Sandbox posture:' line at boot. None means WorkerHangfireRegistrar.ApplyHangfire " + + "no longer calls LocalProcessRunner.LogSandboxPosture (or calls it after an early return); more than one means it is called twice."); + + line.Level.ShouldBe(LogEventLevel.Information); + new[] { "BubblewrapConfines", "BubblewrapUnavailableReason", "McpProxyPresent", "CanSeal", "SealUnavailableReason" }.Except(line.Properties.Keys).ShouldBeEmpty( + customMessage: "the posture line must carry each probe a launch reads, by the property names operators filter on"); + } + + private static bool IsSandboxPostureLine(LogEvent logEvent) => logEvent.MessageTemplate.Text.StartsWith("Sandbox posture:", StringComparison.Ordinal); + /// /// A storage handle the test OWNS, built over the fixture's own connection string with the production options /// (). Identity is therefore by construction: every read, @@ -282,13 +312,15 @@ private static void AssertNothingRolledBack(CapturedLogLines sink) /// One triggered execution: which schedule asked for it, which background job carried it, and the last state observed. private sealed record TickOutcome(string RecurringJobId, string BackgroundJobId, StateData? State); - /// Serilog sink that keeps every rendered line the host writes at Warning or above, so the test can assert on what the pipeline logged. + /// Serilog sink that keeps every event the host's logger lets through (see RecurringJobWorkerHostFactory.CreateHost), so the test can assert on what the pipeline logged. private sealed class CapturedLogLines : ILogEventSink { - private readonly ConcurrentQueue _lines = new(); + private readonly ConcurrentQueue _events = new(); + + public void Emit(LogEvent logEvent) => _events.Enqueue(logEvent); - public void Emit(LogEvent logEvent) => _lines.Enqueue(logEvent.RenderMessage()); + public IReadOnlyList Events() => _events.ToArray(); - public IReadOnlyList Lines() => _lines.ToArray(); + public IReadOnlyList Lines() => _events.Select(logEvent => logEvent.RenderMessage()).ToList(); } } diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/RootlessWorkerPostureTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/RootlessWorkerPostureTests.cs new file mode 100644 index 000000000..923b2204e --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Workflows/RootlessWorkerPostureTests.cs @@ -0,0 +1,169 @@ +using System.Security.Cryptography; +using System.Text; +using System.Text.Json; +using CodeSpace.Core.Services.Agents.Sandbox.Isolation; +using CodeSpace.Core.Services.Agents.Sandbox.Runners; +using Microsoft.Extensions.Logging; +using Shouldly; + +namespace CodeSpace.UnitTests.Workflows; + +/// +/// Pins the rootless worker posture: the committed seccomp profile (backend/deploy/seccomp/codespace-worker.json) +/// that lets the non-root, capability-less worker build bubblewrap's sandbox, and the boot line that tells an operator +/// which parts of that posture hold. That the profile really confines on a kernel is shown by running the worker image +/// under it as uid 1654; this tier pins what the file grants, so a narrowing or a widening is a reviewed diff rather +/// than a surprise on the next deploy. +/// +[Trait("Category", "Unit")] +public sealed class RootlessWorkerPostureTests +{ + /// + /// Every call moby v20.10.20's default allows only to a container holding a capability (the ptrace family, module + /// loading, the new mount API, open_by_handle_at, bpf, perf_event_open and the rest), less the + /// seven bubblewrap needs. The worker holds no capability, so none of these may be allowed. ptrace itself is + /// not here: moby allows it to every container on kernel 4.8+. + /// + private static readonly string[] CapabilityGatedSyscalls = + { + "acct", "bpf", "chroot", "clock_settime", "delete_module", "fanotify_init", "finit_module", "fsconfig", "fsmount", "fsopen", "fspick", "get_mempolicy", "init_module", + "ioperm", "iopl", "kcmp", "lookup_dcookie", "mbind", "mount_setattr", "move_mount", "name_to_handle_at", "open_by_handle_at", "open_tree", "perf_event_open", + "pidfd_getfd", "process_madvise", "process_vm_readv", "process_vm_writev", "quotactl", "quotactl_fd", "reboot", "set_mempolicy", "setdomainname", "sethostname", + "settimeofday", "stime", "syslog", "umount", "vhangup", + }; + + /// The fields of the OCI runtime-spec LinuxSeccomp struct: all containerd keeps when it decodes a Localhost profile. + private static readonly string[] OciProfileKeys = { "defaultAction", "defaultErrnoRet", "architectures", "flags", "listenerPath", "listenerMetadata", "syscalls" }; + + /// The fields of the OCI LinuxSyscall struct, plus comment, which Docker, containerd and CRI-O all ignore. + private static readonly string[] OciRuleKeys = { "names", "action", "errnoRet", "args", "comment" }; + + /// SHA-256 of codespace-worker.json as derive-codespace-worker.sh writes it, with LF line endings. + private const string ReviewedProfileSha256 = "40a59e30e31b4e74b1db959883e0f5c2c678753f4374f09e67755b82fcd55844"; + + /// What bubblewrap calls to build its sandbox, which moby's default profile reserves for CAP_SYS_ADMIN (or, for pivot_root, denies outright). + [Theory] + [InlineData("clone")] + [InlineData("clone3")] + [InlineData("mount")] + [InlineData("pivot_root")] + [InlineData("setns")] + [InlineData("umount2")] + [InlineData("unshare")] + public void The_worker_profile_allows_what_bubblewrap_needs_without_a_capability(string syscall) + { + var rules = ProfileRules().Where(rule => Names(rule).Contains(syscall)).ToList(); + + rules.ShouldContain(rule => IsUnconditionalAllow(rule), $"{syscall} must be allowed for a worker with no capabilities; without it bubblewrap cannot build its sandbox and every run is unconfined"); + rules.ShouldAllBe(rule => Action(rule) == "SCMP_ACT_ALLOW", $"a rule that denies {syscall} beside the one that allows it leaves the outcome to the runtime's rule merge"); + } + + [Fact] + public void The_worker_profile_still_denies_what_it_does_not_list() => + ReadProfile().GetProperty("defaultAction").GetString().ShouldBe("SCMP_ACT_ERRNO", "the profile is moby's default plus bubblewrap's calls, not an allow-list turned inside out"); + + [Fact] + public void The_worker_profile_allows_nothing_else_moby_reserves_for_a_capability() + { + var allowed = ProfileRules().Where(rule => Action(rule) == "SCMP_ACT_ALLOW").SelectMany(Names); + + allowed.Intersect(CapabilityGatedSyscalls).ShouldBeEmpty("only bubblewrap's calls leave moby's capability-gated set; the worker holds no capability, so the rest stay denied"); + } + + [Fact] + public void Every_runtime_reads_the_worker_profile_alike() + { + var profileKeys = ReadProfile().EnumerateObject().Select(property => property.Name); + var ruleKeys = ProfileRules().SelectMany(rule => rule.EnumerateObject()).Select(property => property.Name).Distinct(); + + profileKeys.Except(OciProfileKeys).ShouldBeEmpty("containerd decodes a Localhost profile into the OCI LinuxSeccomp struct and drops any other key (moby's archMap among them), so the file must say everything in OCI's fields"); + ruleKeys.Except(OciRuleKeys).ShouldBeEmpty("containerd drops a rule's includes and excludes, so a rule moby gates on a capability becomes an unconditional allow there; the profile must carry no condition only some runtimes evaluate"); + } + + [Fact] + public void The_worker_profile_is_the_reviewed_derivation() + { + var actual = Convert.ToHexStringLower(SHA256.HashData(Encoding.UTF8.GetBytes(File.ReadAllText(ProfilePath()).ReplaceLineEndings("\n")))); + + actual.ShouldBe(ReviewedProfileSha256, "codespace-worker.json is the output of backend/deploy/seccomp/derive-codespace-worker.sh; change what the worker may call there, re-run it, and update this pin in the same diff"); + } + + [Theory] + [InlineData(true)] + [InlineData(false)] + public void The_boot_posture_line_names_each_probe_a_launch_reads(bool helperPresent) + { + // /bin/sh stands in for a present helper on the POSIX hosts the suite runs on; Windows gets any present file. + var helperPath = helperPresent ? (OperatingSystem.IsWindows() ? Environment.ProcessPath! : "/bin/sh") : "/nonexistent/codespace-mcp"; + var logger = new CapturingLogger(); + + LocalProcessRunner.LogSandboxPosture(logger, helperPath); + + var line = logger.Entries.ShouldHaveSingleItem("the posture is one line, so an operator greps one line"); + line.Level.ShouldBe(LogLevel.Information); + line.Properties["BubblewrapConfines"].ShouldBe(BubblewrapSandbox.Available is not null); + line.Properties["BubblewrapUnavailableReason"].ShouldBe(BubblewrapSandbox.UnavailableReason ?? "none"); + line.Properties["McpProxyPresent"].ShouldBe(helperPresent); + line.Properties["McpProxyPath"].ShouldBe(helperPath); + line.Properties["CanSeal"].ShouldBe(FilteredEgressNetns.CanSeal); + line.Properties["SealUnavailableReason"].ShouldBe(FilteredEgressNetns.SealUnavailableReason ?? "none"); + } + + /// Allowed for every caller on every architecture: no argument filter, and no capability, architecture or kernel condition. + private static bool IsUnconditionalAllow(JsonElement rule) => + Action(rule) == "SCMP_ACT_ALLOW" && IsEmpty(rule, "args") && IsEmpty(rule, "includes") && IsEmpty(rule, "excludes"); + + private static bool IsEmpty(JsonElement rule, string property) + { + if (!rule.TryGetProperty(property, out var value) || value.ValueKind == JsonValueKind.Null) return true; + + return value.ValueKind == JsonValueKind.Array ? value.GetArrayLength() == 0 : !value.EnumerateObject().Any(); + } + + private static string? Action(JsonElement rule) => rule.GetProperty("action").GetString(); + + private static IEnumerable Names(JsonElement rule) => rule.GetProperty("names").EnumerateArray().Select(name => name.GetString()!); + + private static IReadOnlyList ProfileRules() => ReadProfile().GetProperty("syscalls").EnumerateArray().ToList(); + + private static JsonElement ReadProfile() + { + using var profile = JsonDocument.Parse(File.ReadAllText(ProfilePath())); + + return profile.RootElement.Clone(); + } + + private static string ProfilePath() => LocateRepoFile("backend", "deploy", "seccomp", "codespace-worker.json"); + + private static string LocateRepoFile(params string[] segments) + { + var relative = Path.Combine(segments); + + for (var dir = new DirectoryInfo(AppContext.BaseDirectory); dir is not null; dir = dir.Parent) + { + var candidate = Path.Combine(dir.FullName, relative); + if (File.Exists(candidate)) return candidate; + } + + throw new FileNotFoundException($"{relative} not found walking up from {AppContext.BaseDirectory}"); + } + + private sealed class CapturingLogger : ILogger + { + public List<(LogLevel Level, IReadOnlyDictionary Properties)> Entries { get; } = new(); + + public IDisposable BeginScope(TState state) where TState : notnull => NullScope.Instance; + + public bool IsEnabled(LogLevel logLevel) => true; + + public void Log(LogLevel logLevel, EventId eventId, TState state, Exception? exception, Func formatter) => + Entries.Add((logLevel, ((IEnumerable>)state!).ToDictionary(pair => pair.Key, pair => pair.Value))); + + private sealed class NullScope : IDisposable + { + public static readonly NullScope Instance = new(); + + public void Dispose() { } + } + } +} diff --git a/docker-compose.yml b/docker-compose.yml index 47b790620..3108a9396 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -89,9 +89,11 @@ services: # # SAFETY: read-only tools are served UNCONDITIONALLY (McpRequestHandler short-circuits IsReadOnly BEFORE the # governance ledger); side-effecting tools route through the exactly-once ToolCallLedger + AgentToolGate matrix - # (fail-closed). CONFINEMENT: bubblewrap is in the image, but Sandbox:RequireConfinement is NOT armed here — running - # bubblewrap inside an unprivileged container needs userns/privileged settings, so a prod k8s pod that grants - # user namespaces sets Sandbox__RequireConfinement=true (fail-closed); local dev runs agents unconfined-in-container. + # (fail-closed). CONFINEMENT: bubblewrap is in the image, but Docker's default seccomp and AppArmor profiles deny it + # the user namespaces and mounts it needs, and Docker's masked /proc blocks the fresh /proc it mounts, so by default + # local dev runs agents unconfined-in-container and Sandbox:RequireConfinement is NOT armed. The commented + # security_opt block below grants all three without --privileged or any capability (Dockerfile.worker has the same + # posture for Kubernetes). worker: profiles: ["app"] build: @@ -99,6 +101,15 @@ services: dockerfile: Dockerfile.worker container_name: codespace-worker restart: unless-stopped + # Confine agent runs as the image's uid 1654 with no capabilities: uncomment on a host that allows unprivileged + # user namespaces (Compose v2.15+ for systempaths), then arm Sandbox__RequireConfinement below. The worker's boot + # line then reads "Sandbox posture: bubblewrap confines True", but that probe mounts no /proc: without the + # systempaths line it still reads True while every launch fails with bubblewrap's "Can't mount proc on + # /newroot/proc" in the run's stderr, so check that the first run's agent actually starts. + # security_opt: + # - seccomp=./backend/deploy/seccomp/codespace-worker.json + # - apparmor=unconfined + # - systempaths=unconfined depends_on: postgres: condition: service_healthy @@ -125,7 +136,8 @@ services: # The codespace-mcp proxy is published next to the API by the csproj target, so the default path resolves; # set this only to point at an air-gapped mirror or a custom location. CODESPACE_MCP_PROXY_PATH: "${CODESPACE_MCP_PROXY_PATH:-}" - # Prod (userns-capable pod) arms the fail-closed isolation guard; left off for local dev (see comment above). + # Arm the fail-closed isolation guard wherever confinement is granted (the security_opt block above, or a prod pod + # with the same posture), so a lost grant refuses runs instead of running them unconfined; off for local dev. # Sandbox__RequireConfinement: "true" volumes: - codespace-artifacts:/var/lib/codespace/artifacts