From cea111395ddf6c1b7b571cffc86c8cfa236a56b7 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Wed, 16 Sep 2026 12:17:06 +0200 Subject: [PATCH 01/29] Execute Level 2 probes with raw evidence and basic graders --- docs/source/reference/cli.md | 35 ++ pyproject.toml | 1 + src/openenv/cli/commands/validate.py | 22 +- src/openenv/validation/graders/__init__.py | 2 + .../validation/graders/runtime/__init__.py | 5 + .../validation/graders/runtime/basic.py | 195 ++++++++++ src/openenv/validation/runner.py | 290 +++++++++++++-- src/openenv/validation/runtime/artifacts.py | 120 ++++++ src/openenv/validation/runtime/collector.py | 158 ++++++++ src/openenv/validation/runtime/scheduler.py | 82 +++++ .../validation/runtime/schema_worker.py | 44 +++ .../integration/test_runtime_cli.py | 261 +++++++++++++ .../test_validation/test_runtime_artifacts.py | 224 +++++++++++ .../test_validation/test_runtime_collector.py | 107 ++++++ .../test_validation/test_runtime_execution.py | 283 ++++++++++++++ tests/test_validation/test_runtime_grading.py | 347 ++++++++++++++++++ tests/validation_runtime/uv.lock | 2 + 17 files changed, 2146 insertions(+), 32 deletions(-) create mode 100644 src/openenv/validation/graders/runtime/__init__.py create mode 100644 src/openenv/validation/graders/runtime/basic.py create mode 100644 src/openenv/validation/runtime/artifacts.py create mode 100644 src/openenv/validation/runtime/collector.py create mode 100644 src/openenv/validation/runtime/scheduler.py create mode 100644 src/openenv/validation/runtime/schema_worker.py create mode 100644 tests/test_validation/integration/test_runtime_cli.py create mode 100644 tests/test_validation/test_runtime_artifacts.py create mode 100644 tests/test_validation/test_runtime_collector.py create mode 100644 tests/test_validation/test_runtime_execution.py create mode 100644 tests/test_validation/test_runtime_grading.py diff --git a/docs/source/reference/cli.md b/docs/source/reference/cli.md index 5eadeb876d..31720c5a7f 100644 --- a/docs/source/reference/cli.md +++ b/docs/source/reference/cli.md @@ -35,6 +35,41 @@ openenv import path/to/source --name my_env --output-dir ./envs --env-class MyEn ## `openenv validate` +Run static declaration checks without Docker, or build and probe a package locally: + +```bash +openenv validate ./my_env --level static +openenv validate ./my_env --level runtime --local --output report.json +``` + +Runtime validation requires Docker and a `validation.execution` declaration in +`openenv.yaml`. The declaration points to a bounded JSON replay plan containing +one reset and a sequence of actions. The validator builds an immutable image, +opens one WebSocket session, measures reward values, observation schemas and +state continuity, and removes the container even if a check fails. + +This first runtime slice implements startup, reward, observation and state +checks. Other applicable Level 2 checks appear explicitly as `SKIP`; they make +the result `WARN`, which exits zero and does not mean Level 2 is complete. +`FAIL` exits 1, unsupported package formats exit 2, and internal or policy errors +exit 3. `--level semantic` +includes the available lower-level checks but does not claim semantic execution. +`--skip-build` runs declaration checks and skips runtime execution entirely. +The Docker provider currently supports CPU workloads and `public` network mode; +unsupported network or GPU requirements skip runtime before building. + +Runtime reports use schema version 2 and severity policy v2. Static reports for +v1 manifests retain schema version 1 and policy v1. An explicit v1 policy with a +runtime ceiling is rejected. With `--output report.json`, a sibling +`report.artifacts/` directory contains the replay plan, bounded redacted evidence, +coverage inventory, provider settings, cleanup outcome and checksums. Treat these +as author-validation evidence; this command does not issue certification. + +For a pinned, installed-wheel reproduction of the shared fixture and its fault +cases, follow [the runtime lab](../../../tests/validation_runtime/README.md). +The implementation and remaining check inventory are tracked in +[the Level 2 umbrella](https://github.com/huggingface/OpenEnv/issues/1177). + [[autodoc]] openenv.cli.commands.validate.validate ## `openenv push` diff --git a/pyproject.toml b/pyproject.toml index 1c655286e9..6959aee201 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -34,6 +34,7 @@ dependencies = [ # Web UI dependencies "gradio>=4.0.0", "httpx>=0.28.1", + "jsonschema>=4.20.0", ] [project.optional-dependencies] diff --git a/src/openenv/cli/commands/validate.py b/src/openenv/cli/commands/validate.py index 618d3ab8f2..2d943a123d 100644 --- a/src/openenv/cli/commands/validate.py +++ b/src/openenv/cli/commands/validate.py @@ -118,10 +118,16 @@ def validate( help="Skip the image build; build-dependent checks are SKIPped with a reason", ), ] = False, + local: Annotated[ + bool, + typer.Option("--local", help="Use Docker-local runtime validation explicitly"), + ] = False, policy_version: Annotated[ - str, - typer.Option("--policy", help="Severity policy version to apply"), - ] = "v1", + str | None, + typer.Option( + "--policy", help="Severity policy version (static: v1; runtime: v2)" + ), + ] = None, json_output: Annotated[ bool, typer.Option("--json", help="Print the validation report as JSON"), @@ -188,6 +194,9 @@ def validate( runtime_target = target if runtime_target is not None: + if local: + typer.echo("Error: --local cannot be combined with a running URL", err=True) + raise typer.Exit(EXIT_FAIL) try: report = validate_running_environment(runtime_target, timeout_s=timeout) except ValueError as exc: @@ -221,11 +230,16 @@ def validate( raise typer.Exit(EXIT_UNSUPPORTED) try: + if output is not None and not output.parent.is_dir(): + raise OSError("report output directory does not exist") validation_report = run_validation( package_root, max_level=_LEVELS[level], skip_build=skip_build, - policy=load_policy(policy_version), + policy=load_policy(policy_version) if policy_version else None, + artifacts_dir=( + output.parent / (output.stem + ".artifacts") if output else None + ), ) report_json = write_report(validation_report, output) except (SignatureError, UnsupportedPackageError) as exc: diff --git a/src/openenv/validation/graders/__init__.py b/src/openenv/validation/graders/__init__.py index e96c9116dc..26138e1a21 100644 --- a/src/openenv/validation/graders/__init__.py +++ b/src/openenv/validation/graders/__init__.py @@ -13,6 +13,7 @@ from ..manifest import CapabilitiesSpec, NormalizedManifest from ..providers import RunningSubject from ..report import CheckResult +from ..runtime.contracts import RuntimeEvidence from ..types import Level, ProviderCapability ENTRY_POINT_GROUP = "openenv.validation.graders" @@ -42,6 +43,7 @@ class Subject: image_ref: str | None running: RunningSubject | None outputs_dir: Path + runtime_evidence: RuntimeEvidence | None = None @runtime_checkable diff --git a/src/openenv/validation/graders/runtime/__init__.py b/src/openenv/validation/graders/runtime/__init__.py new file mode 100644 index 0000000000..f713c6a7eb --- /dev/null +++ b/src/openenv/validation/graders/runtime/__init__.py @@ -0,0 +1,5 @@ +"""The first runtime slice: raw reward, observation and state contracts.""" + +from .basic import ObservationSchemaGrader, RewardWellFormedGrader, StateContractGrader + +__all__ = ["ObservationSchemaGrader", "RewardWellFormedGrader", "StateContractGrader"] diff --git a/src/openenv/validation/graders/runtime/basic.py b/src/openenv/validation/graders/runtime/basic.py new file mode 100644 index 0000000000..f76960005b --- /dev/null +++ b/src/openenv/validation/graders/runtime/basic.py @@ -0,0 +1,195 @@ +"""Pure graders over independently collected, uncoerced wire evidence.""" + +import json +import math +import subprocess +import sys +import time +from pathlib import Path + +from ...report import CheckResult +from ...types import CheckStatus, Level + + +def _remote_reference(value) -> bool: + if isinstance(value, dict): + for key, child in value.items(): + if key in {"$ref", "$dynamicRef"} and ( + not isinstance(child, str) or not child.startswith("#") + ): + return True + if _remote_reference(child): + return True + elif isinstance(value, list): + return any(_remote_reference(child) for child in value) + return False + + +class _RuntimeGrader: + level = Level.RUNTIME + requires_capabilities = frozenset() + requires_provider = frozenset() + depends_on = ("runtime.startup",) + + def applies_to(self, manifest) -> bool: + return True + + def run(self, subject) -> CheckResult: + started = time.monotonic() + evidence = subject.runtime_evidence + if evidence is None: + return CheckResult( + check_id=self.check_id, + status=CheckStatus.SKIP, + evidence=["runtime evidence is unavailable"], + duration_s=0, + ) + if not evidence.failure_reason and not any( + row.operation == "step" for row in evidence.exchanges + ): + return CheckResult( + check_id=self.check_id, + status=CheckStatus.SKIP, + evidence=["no step was observed; the episode contract is incomplete"], + duration_s=0, + ) + problems = [] + if evidence.failure_reason: + problems.append(evidence.failure_reason) + try: + problems.extend(self.check(subject, evidence)) + except (ValueError, TypeError, KeyError, RecursionError, OverflowError): + problems.append("malformed runtime evidence") + return CheckResult( + check_id=self.check_id, + status=CheckStatus.FAIL if problems else CheckStatus.PASS, + measured={"exchanges": len(evidence.exchanges)}, + evidence=problems[:20] or ["observed runtime contract holds"], + remediation=( + "Fix the reported runtime contract violations." if problems else None + ), + duration_s=time.monotonic() - started, + ) + + +class RewardWellFormedGrader(_RuntimeGrader): + """Require finite numeric step rewards within the manifest's declared range.""" + + check_id = "runtime.reward_well_formed" + + def check(self, subject, evidence) -> list[str]: + problems = [] + observations = 0 + low, high = subject.manifest.reward.range + for index, exchange in enumerate(evidence.exchanges): + if exchange.operation not in {"reset", "step"}: + continue + observations += 1 + data = json.loads(exchange.response_json)["data"] + if "reward" not in data: + problems.append(f"exchange {index}: missing reward") + continue + reward = data["reward"] + if reward is None and exchange.operation == "reset": + continue + if ( + type(reward) not in (int, float) + or (type(reward) is float and not math.isfinite(reward)) + or not low <= reward <= high + ): + problems.append(f"exchange {index}: reward must be finite and in range") + if observations == 0: + problems.append("no observations were measured") + return problems + + +class ObservationSchemaGrader(_RuntimeGrader): + """Validate raw envelopes and reconstructed observations against the schema.""" + + check_id = "runtime.observation_schema" + + def check(self, subject, evidence) -> list[str]: + if evidence.observation_schema_json is None: + return ["observation schema unavailable"] + schema = json.loads(evidence.observation_schema_json) + # Submitted schemas must not cause host-side HTTP/file retrieval. + if _remote_reference(schema): + return ["observation schema has a non-local reference"] + problems = [] + observations = [] + count = 0 + for index, exchange in enumerate(evidence.exchanges): + if exchange.operation not in {"reset", "step"}: + continue + count += 1 + response = json.loads(exchange.response_json) + data = response["data"] + if ( + response.get("type") != "observation" + or not isinstance(data.get("observation"), dict) + or type(data.get("done")) is not bool + or "reward" not in data + ): + problems.append(f"exchange {index}: malformed observation envelope") + continue + observation = dict(data["observation"]) + observation.update(reward=data["reward"], done=data["done"]) + observations.append({"index": index, "observation": observation}) + if count == 0: + problems.append("no observations were measured") + # A subject-supplied regex or recursive schema can exhaust CPU. Keep all + # schema evaluation in a disposable interpreter with a hard wall deadline. + worker = Path(__file__).resolve().parents[2] / "runtime" / "schema_worker.py" + try: + checked = subprocess.run( + [sys.executable, "-I", str(worker)], + input=json.dumps({"schema": schema, "observations": observations}), + capture_output=True, + text=True, + timeout=5, + env={}, + ) + if checked.returncode: + problems.append( + "observation schema evaluation exceeded its resource budget" + ) + else: + problems.extend(json.loads(checked.stdout)) + except subprocess.TimeoutExpired: + problems.append("observation schema evaluation exceeded its time budget") + return problems + + +class StateContractGrader(_RuntimeGrader): + """Verify episode identity and step counts in the same replayed session.""" + + check_id = "runtime.state_contract" + + def check(self, subject, evidence) -> list[str]: + problems = [] + episode_id = None + steps = 0 + states = 0 + for index, exchange in enumerate(evidence.exchanges): + if exchange.operation == "reset": + episode_id = json.loads(exchange.request_json)["data"]["episode_id"] + steps = 0 + elif exchange.operation == "step": + steps += 1 + elif exchange.operation == "state": + states += 1 + response = json.loads(exchange.response_json) + data = response["data"] + if ( + response.get("type") != "state" + or data.get("episode_id") != episode_id + ): + problems.append(f"exchange {index}: episode_id differs from reset") + if ( + type(data.get("step_count")) is not int + or data["step_count"] != steps + ): + problems.append(f"exchange {index}: incorrect step_count") + if states == 0: + problems.append("no state snapshots were measured") + return problems diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index 7f40e9fc91..f4d24a177d 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -1,18 +1,32 @@ """Validation orchestration: parse → grade → apply policy → report.""" import hashlib +import os +import stat import time +import uuid +from dataclasses import replace from pathlib import Path from .graders import GraderRegistry, Subject +from .graders.runtime import ( + ObservationSchemaGrader, + RewardWellFormedGrader, + StateContractGrader, +) from .graders.static import StaticManifestGrader from .manifest import ManifestError, NormalizedManifest, NormalizedManifestV2 from .parsers import ParserRegistry from .parsers.openenv_yaml import OpenEnvYamlParser -from .policy import apply_policy, load_policy, SeverityPolicy +from .policy import apply_policy, load_policy, PolicyError, SeverityPolicy +from .providers import ProviderError, StartupError, UnsupportedCapability from .report import CheckResult, ValidationReport, ValidationReportV2 +from .runtime.artifacts import write_runtime_bundle +from .runtime.collector import collect_runtime_evidence +from .runtime.contracts import LaunchSpec, load_runtime_plan, RuntimePlanError +from .runtime.scheduler import execute_graders from .signature import detect_signature -from .types import CheckStatus, Lane, Level +from .types import CheckStatus, Lane, Level, ProviderCapability REPORT_SCHEMA_VERSION = "1" @@ -34,15 +48,22 @@ def source_digest(package_root: Path) -> str: files = [] for path in package_root.rglob("*"): relative_path = path.relative_to(package_root) - if path.is_file() and not any( - part in _DIGEST_EXCLUDED_DIRS for part in relative_path.parts - ): + if any(part in _DIGEST_EXCLUDED_DIRS for part in relative_path.parts): + continue + if path.is_symlink(): + raise ValueError("validation source may not contain symbolic links") + if path.is_file(): files.append((relative_path.as_posix(), path)) for relative_path, path in sorted(files, key=lambda item: item[0]): digest.update(relative_path.encode()) digest.update(b"\0") - digest.update(path.read_bytes()) + fd = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + with os.fdopen(fd, "rb") as source: + if not stat.S_ISREG(os.fstat(source.fileno()).st_mode): + raise ValueError("validation source must contain regular files") + while chunk := source.read(1024 * 1024): + digest.update(chunk) digest.update(b"\0") return digest.hexdigest() @@ -54,17 +75,170 @@ def default_parser_registry() -> ParserRegistry: return registry -def _run_grader(grader, subject: Subject) -> CheckResult: +def _outcome(check_id, status, reason, *, started=None, measured=None): + return CheckResult( + check_id=check_id, + status=status, + evidence=[reason], + measured=measured or {}, + duration_s=time.monotonic() - started if started is not None else 0, + ) + + +def _applicable(check_id, manifest): + if manifest is None: + return True + if check_id in {"runtime.rubric_introspectable", "runtime.reward_attribution"}: + return manifest.capabilities.rubric_tree + if check_id == "runtime.task_declaration_accuracy": + return manifest.capabilities.task_api or bool( + manifest.capabilities.declared_task_count + ) + return True + + +def _runtime(subject, *, skip_build, provider): + """Own build/start/collection/stop and retain failure evidence through teardown.""" + manifest = subject.manifest + plan = None + evidence = None + running = None + inspection = {} + cleanup = {"required": False, "completed": True} started = time.monotonic() + attempted = False + result = None try: - return grader.run(subject) + if skip_build: + raise UnsupportedCapability( + "--skip-build: build-dependent runtime checks skipped" + ) + if not isinstance(manifest, NormalizedManifestV2): + raise UnsupportedCapability( + "missing validation.execution declaration and runtime plan" + ) + plan = load_runtime_plan(subject.root, manifest.execution) + if provider is None: + from .providers.docker import DockerValidationProvider + + provider = DockerValidationProvider() + if manifest.network.mode not in provider.supported_network_modes: + raise UnsupportedCapability( + f"provider cannot enforce network mode {manifest.network.mode}" + ) + if ( + manifest.resources.gpus + and ProviderCapability.GPU not in provider.capabilities + ): + raise UnsupportedCapability("missing provider capability: gpu") + if ProviderCapability.IMAGE_BUILD not in provider.capabilities: + raise UnsupportedCapability("missing provider capability: image_build") + # Validate all launch constraints before any container or build is started. + spec = LaunchSpec( + image_ref="sha256:" + "0" * 64, + resources=manifest.resources, + network=manifest.network, + run_id="validation-" + uuid.uuid4().hex, + ) + attempted = True + image_ref = provider.build(subject.root, manifest.execution) + running = provider.start( + LaunchSpec.model_validate({**spec.model_dump(), "image_ref": image_ref}) + ) + cleanup = {"required": True, "completed": False} + inspection = running.inspect() + result = _outcome( + "runtime.startup", + CheckStatus.PASS, + "subject built and reached its control endpoint", + started=started, + measured={"provider": provider.name, "image_ref": image_ref}, + ) + evidence = collect_runtime_evidence( + running.base_url, + plan, + episode_timeout_s=min(manifest.resources.episode_timeout_s, 300.0), + ) + # A health endpoint without a functioning protocol isn't a startup success. + if not evidence.exchanges and evidence.failure_reason: + result = _outcome( + "runtime.startup", + CheckStatus.FAIL, + evidence.failure_reason, + started=started, + ) + subject = replace( + subject, image_ref=image_ref, running=running, runtime_evidence=evidence + ) + checks = execute_graders( + [ + RewardWellFormedGrader(), + ObservationSchemaGrader(), + StateContractGrader(), + ], + subject, + provider_capabilities=provider.capabilities, + prior=[result], + ) + except UnsupportedCapability as exc: + result = _outcome( + "runtime.startup", CheckStatus.SKIP, str(exc), started=started + ) + checks = [] + except RuntimePlanError: + result = _outcome( + "runtime.startup", + CheckStatus.FAIL, + "runtime plan is missing, unsafe or invalid", + started=started, + ) + checks = [] + except StartupError: + result = _outcome( + "runtime.startup", + CheckStatus.FAIL, + "subject failed build or readiness; inspect the Docker fixture/build inputs", + started=started, + ) + checks = [] + except ProviderError: + result = _outcome( + "runtime.startup", + CheckStatus.ERROR, + "provider could not complete a bounded operation", + started=started, + ) + checks = [] + except KeyboardInterrupt: + result = _outcome( + "runtime.startup", + CheckStatus.ERROR, + "validation interrupted", + started=started, + ) + checks = [] except Exception as exc: - return CheckResult( - check_id=grader.check_id, - status=CheckStatus.ERROR, - evidence=[f"grader crashed: {exc!r}"], - duration_s=time.monotonic() - started, + result = _outcome( + "runtime.startup", + CheckStatus.ERROR, + f"runtime orchestration failed ({type(exc).__name__})", + started=started, ) + checks = [] + finally: + if running is not None: + try: + running.stop() + cleanup["completed"] = True + except Exception: + cleanup["completed"] = False + result = _outcome( + "runtime.startup", + CheckStatus.ERROR, + "subject teardown failed", + started=started, + ) + return [result, *checks], attempted, plan, evidence, inspection, cleanup def run_validation( @@ -73,7 +247,9 @@ def run_validation( max_level: Level = Level.SEMANTIC, skip_build: bool = False, policy: SeverityPolicy | None = None, -) -> ValidationReport: + provider=None, + artifacts_dir: Path | None = None, +) -> ValidationReport | ValidationReportV2: """ Validate a package end to end and return the report. @@ -92,15 +268,24 @@ def run_validation( skip_build (`bool`, *optional*, defaults to `False`): Skip the image build; build-dependent checks SKIP with a reason. policy ([`~openenv.validation.policy.SeverityPolicy`], *optional*): - Severity policy; `None` loads the committed default version. + `None` chooses v1 for static and v2 for runtime/semantic ceilings. + provider ([`~openenv.validation.providers.ValidationProvider`], *optional*): + Validation-only provider; defaults to Docker-local for runtime runs. + artifacts_dir (`Path`, *optional*): + Write a redacted reproduction bundle to this directory. Returns: [`~openenv.validation.report.ValidationReport`]: the completed report. """ - del skip_build target = Path(target) - policy = policy or load_policy() + wants_runtime = max_level >= Level.RUNTIME + policy = policy or load_policy("v2" if wants_runtime else "v1") + if wants_runtime and "runtime.startup" not in policy.entries_for_lane(Lane.LOCAL): + raise PolicyError( + "runtime validation requires policy v2; v1 supports --level static" + ) signature = detect_signature(target) + digest_before = source_digest(target) parser = default_parser_registry().parser_for(signature) manifest: NormalizedManifest | None = None @@ -121,6 +306,10 @@ def run_validation( ) ) + levels = [Level.STATIC] + plan = evidence = None + inspection = {} + cleanup = {"required": False, "completed": True} if manifest is not None: graders = GraderRegistry() graders.register(StaticManifestGrader(policy.bounds)) @@ -131,27 +320,72 @@ def run_validation( running=None, outputs_dir=target / "outputs", ) - results.extend( - _run_grader(grader, subject) - for grader in graders.select(manifest, max_level) - ) + results.extend(execute_graders(graders.select(manifest, Level.STATIC), subject)) + if wants_runtime: + runtime_results, attempted, plan, evidence, inspection, cleanup = _runtime( + subject, skip_build=skip_build, provider=provider + ) + results.extend(runtime_results) + if attempted: + levels.append(Level.RUNTIME) + + if wants_runtime: + # Policy IDs are an inventory, not evidence that a grader exists. Keep the + # incomplete surface explicit throughout the staged implementation. + present = {result.check_id for result in results} + for entry in policy.entries_for_lane(Lane.LOCAL).values(): + if ( + entry.level in {Level.RUNTIME, Level.SEMANTIC} + and entry.level <= max_level + and entry.check_id not in present + and _applicable(entry.check_id, manifest) + ): + reason = "grader not implemented in this build" + if manifest is None: + reason = "unmet dependency: valid manifest" + elif entry.check_id in { + "runtime.reward_well_formed", + "runtime.observation_schema", + "runtime.state_contract", + }: + reason = "unmet dependency: runtime.startup" + results.append(_outcome(entry.check_id, CheckStatus.SKIP, reason)) + if source_digest(target) != digest_before: + results = [r for r in results if r.check_id != "runtime.startup"] + results.append( + _outcome( + "runtime.startup", + CheckStatus.ERROR, + "package source changed during validation", + ) + ) report_type = ( ValidationReportV2 - if isinstance(manifest, NormalizedManifestV2) + if wants_runtime or isinstance(manifest, NormalizedManifestV2) else ValidationReport ) - return report_type( - report_schema_version=( - "2" if report_type is ValidationReportV2 else REPORT_SCHEMA_VERSION - ), + report = report_type( + report_schema_version="2" + if report_type is ValidationReportV2 + else REPORT_SCHEMA_VERSION, target=str(target), - source_digest=source_digest(target), + source_digest=digest_before, signature=signature, manifest=manifest, policy_version=policy.policy_version, lane=Lane.LOCAL, - levels_run=[Level.STATIC], + levels_run=levels, results=results, verdict=apply_policy(results, policy, Lane.LOCAL), ) + if artifacts_dir is not None: + write_runtime_bundle( + artifacts_dir, + report, + plan=plan, + evidence=evidence, + provider=inspection, + cleanup=cleanup, + ) + return report diff --git a/src/openenv/validation/runtime/artifacts.py b/src/openenv/validation/runtime/artifacts.py new file mode 100644 index 0000000000..d2d8e340bb --- /dev/null +++ b/src/openenv/validation/runtime/artifacts.py @@ -0,0 +1,120 @@ +"""Small, bounded reproduction bundles; reports remain the public result contract.""" + +import hashlib +import json +import math +import platform +import re +from dataclasses import asdict +from pathlib import Path + +_SECRET_KEY = re.compile(r"(?i)(password|secret|token|authorization|api[_-]?key)") +_TOKEN = re.compile( + r"(?:hf_[A-Za-z0-9]{8,}|(?:sk|ghp|github_pat)[-_][A-Za-z0-9_-]{8,}|(?i:bearer)\s+\S+)" +) +_MAX_ARTIFACT_DEPTH = 64 + + +def _redact(value, *, depth=0): + if depth >= _MAX_ARTIFACT_DEPTH and isinstance(value, (dict, list)): + return {"omitted": "artifact nesting limit exceeded"} + if isinstance(value, dict): + return { + key: "[REDACTED]" + if _SECRET_KEY.search(key) + else _redact(child, depth=depth + 1) + for key, child in value.items() + } + if isinstance(value, list): + return [_redact(child, depth=depth + 1) for child in value] + if isinstance(value, str): + return _TOKEN.sub("[REDACTED]", value) + if isinstance(value, float) and not math.isfinite(value): + return {"invalid_number": str(value)} + return value + + +def write_runtime_bundle( + directory: Path, report, *, plan=None, evidence=None, provider=None, cleanup=None +): + """Write a credential-filtered trace, source/policy/plan provenance and coverage.""" + directory.mkdir(parents=True, exist_ok=True) + files = {} + coverage = { + "requested_runtime_checks": [ + r.check_id for r in report.results if r.check_id.startswith("runtime.") + ], + "executed": [r.check_id for r in report.results if r.status.value != "skip"], + "incomplete": [r.check_id for r in report.results if r.status.value == "skip"], + } + files["coverage.json"] = coverage + files["report.json"] = report.model_dump(mode="json") + files["cleanup.json"] = cleanup or {"required": False} + if plan: + files["runtime-plan.json"] = plan.model_dump(mode="json") + files["run-manifest.json"] = { + "bundle_schema_version": "1", + "source_digest": report.source_digest, + "policy_version": report.policy_version, + "manifest_schema_version": report.manifest.manifest_schema_version + if report.manifest + else None, + "plan_digest": hashlib.sha256(plan.model_dump_json().encode()).hexdigest() + if plan + else None, + "plan_redacted": bool( + plan + and _redact(plan.model_dump(mode="json")) != plan.model_dump(mode="json") + ), + "platform": platform.platform(), + "python": platform.python_version(), + "provider": provider or {}, + } + trace = [] + collector_metadata = None + if evidence: + omitted_fields = [] + for index, exchange in enumerate(evidence.exchanges): + row = asdict(exchange) + for key in ("request_json", "response_json"): + try: + row[key] = json.loads(row[key]) + except (ValueError, RecursionError): + row[key] = "[malformed response omitted]" + omitted_fields.append({"exchange_index": index, "field": key}) + trace.append(row) + schema = None + schema_parse_failed = False + if evidence.observation_schema_json is not None: + try: + schema = json.loads(evidence.observation_schema_json) + except (ValueError, RecursionError): + schema_parse_failed = True + collector_metadata = { + "evidence_schema_version": "1", + "trace_file": "collector-trace.json", + "observation_schema": schema, + "schema_available": evidence.observation_schema_json is not None, + "schema_parse_failed": schema_parse_failed, + "omitted_trace_fields": omitted_fields, + "failure_phase": evidence.failure_phase, + "failure_reason": evidence.failure_reason, + "complete": evidence.failure_phase is None + and evidence.failure_reason is None, + } + collector_metadata["redacted"] = ( + bool(omitted_fields) + or schema_parse_failed + or _redact(trace) != trace + or _redact(collector_metadata) != collector_metadata + ) + files["collector-trace.json"] = trace + files["collector-evidence.json"] = collector_metadata + digests = [] + for name, value in files.items(): + payload = ( + json.dumps(_redact(value), indent=2, sort_keys=True, allow_nan=False) + "\n" + ).encode() + (directory / name).write_bytes(payload) + digests.append(f"{hashlib.sha256(payload).hexdigest()} {name}\n") + (directory / "SHA256SUMS").write_text("".join(digests)) diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py new file mode 100644 index 0000000000..f3cf318a98 --- /dev/null +++ b/src/openenv/validation/runtime/collector.py @@ -0,0 +1,158 @@ +"""Bounded raw protocol collection in one OpenEnv orchestration session.""" + +import json +import time +from urllib.parse import urlsplit, urlunsplit + +import httpx +from websockets.sync.client import connect + +from .contracts import RuntimeEvidence, RuntimePlan, WireExchange + +MAX_MESSAGE_BYTES = 1024 * 1024 +MAX_TRACE_BYTES = 8 * 1024 * 1024 + + +def collect_runtime_evidence( + base_url: str, + plan: RuntimePlan, + *, + episode_timeout_s: float, + request_timeout_s: float = 5.0, +) -> RuntimeEvidence: + """ + Preserve schema and reset/step/state responses without model coercion. + + The collector never imports submitted code or reconnects midway through an + episode. Transport failures retain the completed prefix and a bounded reason. + Graders inspect the original envelopes rather than convenience-client defaults. + + Args: + base_url (`str`): + Provider-owned control endpoint. + plan ([`~openenv.validation.runtime.contracts.RuntimePlan`]): + Validated, bounded reset and action inputs. + episode_timeout_s (`float`): + Deadline for the complete collection, including schema retrieval. + request_timeout_s (`float`, *optional*, defaults to `5.0`): + Per-operation deadline, capped by the remaining episode budget. + + Returns: + [`~openenv.validation.runtime.contracts.RuntimeEvidence`]: raw evidence. + """ + deadline = time.monotonic() + episode_timeout_s + exchanges = [] + schema_json = None + phase = "schema" + trace_bytes = 0 + + def remaining() -> float: + value = min(request_timeout_s, deadline - time.monotonic()) + if value <= 0: + raise TimeoutError("episode deadline exceeded") + return value + + try: + with httpx.Client(trust_env=False, follow_redirects=False) as client: + with client.stream( + "GET", + base_url.rstrip("/") + "/schema", + timeout=remaining(), + headers={"Accept-Encoding": "identity"}, + ) as response: + response.raise_for_status() + # iter_bytes() transparently decompresses. Reject compressed + # bodies before touching the stream so the byte budget also + # bounds allocation, even when a subject ignores our header. + if ( + response.headers.get("Content-Encoding", "identity").strip().lower() + != "identity" + ): + raise ValueError("compressed schema responses are not supported") + payload = bytearray() + for chunk in response.iter_bytes(): + remaining() + if len(payload) + len(chunk) > MAX_MESSAGE_BYTES: + raise ValueError("schema exceeds size bound") + payload.extend(chunk) + schema = json.loads(payload) + if not isinstance(schema, dict) or "observation" not in schema: + raise ValueError("missing observation schema") + schema_json = json.dumps(schema["observation"], allow_nan=False) + + endpoint = urlsplit(base_url) + ws_url = urlunsplit( + ( + "wss" if endpoint.scheme == "https" else "ws", + endpoint.netloc, + endpoint.path.rstrip("/") + "/ws", + "", + "", + ) + ) + phase = "connect" + with connect( + ws_url, + proxy=None, + open_timeout=remaining(), + close_timeout=1, + max_size=MAX_MESSAGE_BYTES, + max_queue=1, + compression=None, + ) as socket: + + def exchange(operation: str, data: dict | None = None) -> dict: + nonlocal phase, trace_bytes + phase = operation + request = {"type": operation} + if data is not None: + request["data"] = data + request_json = json.dumps(request, allow_nan=False) + remaining() + socket.send(request_json) + raw = socket.recv(timeout=remaining()) + if not isinstance(raw, str): + raise ValueError("binary response is not the JSON protocol") + trace_bytes += len(raw.encode("utf-8")) + len( + request_json.encode("utf-8") + ) + if trace_bytes > MAX_TRACE_BYTES: + raise ValueError("trace exceeds total size bound") + exchanges.append( + WireExchange( + operation=operation, + request_json=request_json, + response_json=raw, + ) + ) + response = json.loads(raw) + expected = "state" if operation == "state" else "observation" + if ( + not isinstance(response, dict) + or response.get("type") != expected + or not isinstance(response.get("data"), dict) + ): + raise ValueError("unexpected response envelope") + return response["data"] + + reset = dict(plan.reset.options) + reset.update(seed=plan.reset.seed, episode_id=plan.reset.episode_id) + observation = exchange("reset", reset) + exchange("state") + for action in plan.actions: + if observation.get("done") is True: + break + observation = exchange("step", action) + exchange("state") + socket.send(json.dumps({"type": "close"})) + return RuntimeEvidence( + exchanges=tuple(exchanges), observation_schema_json=schema_json + ) + except Exception as exc: + # Exception text may include submitted payloads or URL credentials. + return RuntimeEvidence( + exchanges=tuple(exchanges), + observation_schema_json=schema_json, + failure_phase=phase, + failure_reason=f"{phase} failed ({type(exc).__name__})", + ) diff --git a/src/openenv/validation/runtime/scheduler.py b/src/openenv/validation/runtime/scheduler.py new file mode 100644 index 0000000000..012cabba17 --- /dev/null +++ b/src/openenv/validation/runtime/scheduler.py @@ -0,0 +1,82 @@ +"""Deterministic dependency ordering and capability-aware grader execution.""" + +import time + +from ..policy import PolicyError +from ..report import CheckResult +from ..types import CheckStatus + + +def order_graders(graders: list) -> list: + """Topologically order selected graders; reject duplicates and dependency cycles.""" + by_id = {grader.check_id: grader for grader in graders} + if len(by_id) != len(graders): + raise PolicyError("duplicate grader IDs") + visiting = set() + visited = set() + ordered = [] + + def visit(check_id): + if check_id in visiting: + raise PolicyError(f"grader dependency cycle at {check_id}") + if check_id in visited: + return + visiting.add(check_id) + grader = by_id[check_id] + for dependency in sorted(grader.depends_on): + if dependency in by_id: + visit(dependency) + visiting.remove(check_id) + visited.add(check_id) + ordered.append(grader) + + for check_id in sorted(by_id): + visit(check_id) + return ordered + + +def execute_graders(graders, subject, *, provider_capabilities=frozenset(), prior=()): + """Run independent checks despite failures, and name unmet dependencies in SKIPs.""" + results = [] + outcomes = {result.check_id: result for result in prior} + for grader in order_graders(graders): + blocked = [ + name + for name in grader.depends_on + if name not in outcomes or outcomes[name].status is not CheckStatus.PASS + ] + missing = sorted( + capability.value + for capability in grader.requires_provider - provider_capabilities + ) + if blocked or missing: + reasons = [] + if blocked: + reasons.append("unmet dependencies: " + ", ".join(blocked)) + if missing: + reasons.append("missing provider capabilities: " + ", ".join(missing)) + result = CheckResult( + check_id=grader.check_id, + status=CheckStatus.SKIP, + evidence=reasons, + duration_s=0, + ) + else: + started = time.monotonic() + try: + result = grader.run(subject) + if ( + not isinstance(result, CheckResult) + or result.check_id != grader.check_id + ): + raise ValueError("grader returned the wrong check result") + except Exception as exc: + result = CheckResult( + check_id=grader.check_id, + status=CheckStatus.ERROR, + evidence=[f"grader failed ({type(exc).__name__})"], + duration_s=time.monotonic() - started, + ) + outcomes[grader.check_id] = result + results.append(result) + return results diff --git a/src/openenv/validation/runtime/schema_worker.py b/src/openenv/validation/runtime/schema_worker.py new file mode 100644 index 0000000000..90251ce2fe --- /dev/null +++ b/src/openenv/validation/runtime/schema_worker.py @@ -0,0 +1,44 @@ +"""Disposable JSON Schema evaluator; invoked as a script to avoid package imports.""" + +import json +import sys + +from jsonschema import Draft202012Validator +from referencing import Registry + + +def main(): + """Evaluate bounded inputs, printing only error locations, never subject values.""" + try: + import resource + + resource.setrlimit(resource.RLIMIT_CPU, (3, 4)) + if sys.platform == "linux": + resource.setrlimit(resource.RLIMIT_AS, (512 * 1024 * 1024,) * 2) + except (ImportError, ValueError, OSError): + pass # The parent always enforces the independent wall-clock deadline. + payload = json.loads(sys.stdin.read(10 * 1024 * 1024)) + problems = [] + try: + schema = payload["schema"] + Draft202012Validator.check_schema(schema) + # An empty registry has no retrieval callback: unresolved references + # cannot trigger host filesystem or network access. + validator = Draft202012Validator(schema, registry=Registry()) + for row in payload["observations"]: + for error in validator.iter_errors(row["observation"]): + location = "/".join(str(x) for x in error.absolute_path)[:160] + problems.append( + f"exchange {row['index']}: schema mismatch at {location or '/'}" + ) + if len(problems) >= 20: + break + if len(problems) >= 20: + break + except Exception: + problems.append("invalid or unevaluable observation schema") + sys.stdout.write(json.dumps(problems)) + + +if __name__ == "__main__": + main() diff --git a/tests/test_validation/integration/test_runtime_cli.py b/tests/test_validation/integration/test_runtime_cli.py new file mode 100644 index 0000000000..630e9716a4 --- /dev/null +++ b/tests/test_validation/integration/test_runtime_cli.py @@ -0,0 +1,261 @@ +"""Installed-wheel CLI acceptance: real builds, wire faults, reports and cleanup.""" + +import hashlib +import json +import os +import shutil +import subprocess +import sys +from importlib.resources import files +from pathlib import Path + +import jsonschema +import pytest + + +pytestmark = pytest.mark.docker + +IMPLEMENTED = { + "runtime.startup", + "runtime.reward_well_formed", + "runtime.observation_schema", + "runtime.state_contract", +} +PENDING = { + "runtime.trajectory_record", + "runtime.tool_declaration_accuracy", + "runtime.seed_control", + "runtime.episode_determinism", + "runtime.network_policy", + "runtime.host_containment", + "runtime.resource_bounds", + "runtime.episode_isolation", + "runtime.oracle_containment", +} +NOT_APPLICABLE = { + "runtime.rubric_introspectable", + "runtime.reward_attribution", + "runtime.task_declaration_accuracy", +} + + +def _copy_asset(source, destination): + # Wheel bytes are immutable inputs; hardlinks avoid copying the locked + # wheelhouse for every fault case. Fall back across filesystem boundaries. + try: + os.link(source, destination) + return destination + except OSError: + return shutil.copy2(source, destination) + + +@pytest.fixture +def cli_context(tmp_path): + root = os.environ.get("OPENENV_VALIDATION_CONTEXT") + if not root: + if os.environ.get("OPENENV_REQUIRE_DOCKER") == "1": + pytest.fail("Required CLI acceptance needs OPENENV_VALIDATION_CONTEXT") + pytest.skip("Run the validation lab to provide its offline build context") + context = tmp_path / "subject" + shutil.copytree(root, context, copy_function=_copy_asset) + # Fault injection must never modify the shared context's Dockerfile inode. + dockerfile = context / "Dockerfile" + content = dockerfile.read_bytes() + dockerfile.unlink() + dockerfile.write_bytes(content) + return context + + +def _container_ids(): + checked = subprocess.run( + [ + "docker", + "container", + "ls", + "--all", + "--quiet", + "--filter", + "label=org.openenv.validation.run", + ], + capture_output=True, + text=True, + timeout=10, + check=True, + ) + return set(checked.stdout.split()) + + +def _verify_bundle(bundle, report): + schema = json.loads( + files("openenv.validation") + .joinpath("schemas/report-v2.schema.json") + .read_text() + ) + jsonschema.validate(report, schema) + assert report["report_schema_version"] == "2" + assert report["manifest"]["manifest_schema_version"] == "2" + assert report["policy_version"] == "v2" + assert report["lane"] == "local" + assert len(report["source_digest"]) == 64 + artifact_report = json.loads((bundle / "report.json").read_text()) + assert artifact_report == report + recorded = set() + for line in (bundle / "SHA256SUMS").read_text().splitlines(): + digest, name = line.split(" ", 1) + assert name not in recorded + recorded.add(name) + assert hashlib.sha256((bundle / name).read_bytes()).hexdigest() == digest + assert recorded == { + path.name for path in bundle.iterdir() if path.name != "SHA256SUMS" + } + return {name: json.loads((bundle / name).read_text()) for name in recorded} + + +def _invoke_cli(context, tmp_path, case, *, skip_build=False): + artifact_root = Path(os.environ.get("OPENENV_VALIDATION_ARTIFACTS", tmp_path)) + work = artifact_root / "cli" / case + work.mkdir(parents=True, exist_ok=True) + report_path = work / "report.json" + environment = os.environ.copy() + environment.pop("PYTHONPATH", None) + environment["PYTHONNOUSERSITE"] = "1" + marker = tmp_path / "unexpected-docker-call" + if skip_build: + binary_dir = tmp_path / "bin" + binary_dir.mkdir() + docker = binary_dir / "docker" + docker.write_text( + f"#!{sys.executable}\nfrom pathlib import Path\n" + f"Path({str(marker)!r}).write_text('called')\nraise SystemExit(91)\n" + ) + docker.chmod(0o755) + environment["PATH"] = str(binary_dir) + os.pathsep + environment["PATH"] + command = [ + sys.executable, + "-m", + "openenv.cli", + "validate", + str(context), + "--level", + "runtime", + "--local", + "--policy", + "v2", + "--json", + "--output", + str(report_path), + ] + if skip_build: + command.append("--skip-build") + before = _container_ids() + try: + result = subprocess.run( + command, + cwd=work, + env=environment, + capture_output=True, + text=True, + timeout=180, + ) + finally: + remaining = _container_ids() - before + (work / "cleanup-external.json").write_text( + json.dumps({"remaining_container_ids": sorted(remaining)}) + "\n" + ) + assert not remaining, "CLI validation leaked a container" + (work / "stdout.json").write_text(result.stdout) + (work / "stderr.log").write_text(result.stderr) + assert report_path.is_file(), result.stderr + report = json.loads(report_path.read_text()) + assert json.loads(result.stdout) == report + artifacts = _verify_bundle(work / "report.artifacts", report) + assert not marker.exists(), "--skip-build invoked Docker" + checks = {row["check_id"]: row for row in report["results"]} + assert len(checks) == len(report["results"]), "Report contains duplicate check IDs" + assert { + key for key in checks if key.startswith("runtime.") + } == IMPLEMENTED | PENDING + assert not NOT_APPLICABLE & checks.keys() + assert all(checks[key]["status"] == "skip" for key in PENDING) + assert all(checks[key]["evidence"] for key in PENDING) + assert ( + set(artifacts["coverage.json"]["requested_runtime_checks"]) + == IMPLEMENTED | PENDING + ) + assert artifacts["cleanup.json"]["completed"] is True + return result, report, checks, artifacts + + +@pytest.mark.parametrize( + "mode,failed_check", + [ + ("good", None), + ("bad_reward", "runtime.reward_well_formed"), + ("bad_observation", "runtime.observation_schema"), + ("bad_state", "runtime.state_contract"), + ], +) +def test_cli_runtime_contract_findings(cli_context, tmp_path, mode, failed_check): + with (cli_context / "Dockerfile").open("a") as stream: + stream.write(f"\nENV VALIDATION_FAULT={mode}\n") + result, report, checks, artifacts = _invoke_cli(cli_context, tmp_path, mode) + assert report["levels_run"] == [1, 2] + assert checks["static.manifest"]["status"] == "pass" + assert checks["runtime.startup"]["status"] == "pass" + assert checks["runtime.startup"]["measured"]["image_ref"].startswith("sha256:") + assert artifacts["cleanup.json"]["required"] is True + assert artifacts["run-manifest.json"]["provider"]["container_id"] + assert artifacts["run-manifest.json"]["source_digest"] == report["source_digest"] + trace = artifacts["collector-trace.json"] + assert [row["operation"] for row in trace] == [ + "reset", + "state", + "step", + "state", + "step", + "state", + ] + if failed_check: + assert result.returncode == 1 + assert report["verdict"] == "fail" + assert checks[failed_check]["status"] == "fail" + assert checks[failed_check]["evidence"] + else: + assert result.returncode == 0 + assert report["verdict"] == "warn", ( + "Missing later graders must keep the result incomplete" + ) + assert all(checks[key]["status"] == "pass" for key in IMPLEMENTED) + assert set(artifacts["coverage.json"]["incomplete"]) == PENDING + + +def test_cli_startup_failure_is_a_finding_and_leaves_no_container( + cli_context, tmp_path +): + with (cli_context / "Dockerfile").open("a") as stream: + stream.write("\nENV VALIDATION_FAULT=startup_failure\n") + result, report, checks, artifacts = _invoke_cli( + cli_context, tmp_path, "startup_failure" + ) + assert result.returncode == 1 + assert report["verdict"] == "fail" + assert report["levels_run"] == [1, 2] + assert checks["runtime.startup"]["status"] == "fail" + assert all( + checks[key]["status"] == "skip" for key in IMPLEMENTED - {"runtime.startup"} + ) + assert artifacts["collector-trace.json"] == [] + + +def test_cli_skip_build_does_not_invoke_docker(cli_context, tmp_path): + result, report, checks, artifacts = _invoke_cli( + cli_context, tmp_path, "skip_build", skip_build=True + ) + assert result.returncode == 0 + assert report["verdict"] == "warn" + assert report["levels_run"] == [1] + assert all(checks[key]["status"] == "skip" for key in IMPLEMENTED | PENDING) + assert "--skip-build" in " ".join(checks["runtime.startup"]["evidence"]) + assert artifacts["cleanup.json"]["required"] is False + assert artifacts["run-manifest.json"]["provider"] == {} + assert artifacts["collector-trace.json"] == [] diff --git a/tests/test_validation/test_runtime_artifacts.py b/tests/test_validation/test_runtime_artifacts.py new file mode 100644 index 0000000000..3921447f8a --- /dev/null +++ b/tests/test_validation/test_runtime_artifacts.py @@ -0,0 +1,224 @@ +import hashlib +import json +from dataclasses import replace + +from conftest import load_fixture_manifest +from openenv.validation.graders import Subject +from openenv.validation.graders.runtime import ( + ObservationSchemaGrader, + RewardWellFormedGrader, + StateContractGrader, +) +from openenv.validation.manifest import NormalizedManifest +from openenv.validation.report import ValidationReportV2 +from openenv.validation.runtime.artifacts import write_runtime_bundle +from openenv.validation.runtime.contracts import RuntimeEvidence, WireExchange +from openenv.validation.types import Lane, Level, SignatureKind, Verdict +from support.runtime import evidence, exchange + + +def report(): + return ValidationReportV2( + report_schema_version="2", + target="subject", + source_digest="0" * 64, + signature=SignatureKind.OPENENV_SERVED, + manifest=NormalizedManifest.model_validate( + load_fixture_manifest("served_min_pass") + ), + policy_version="v2", + lane=Lane.LOCAL, + levels_run=[Level.STATIC, Level.RUNTIME], + results=[], + verdict=Verdict.WARN, + ) + + +def measured(): + return evidence( + exchange( + "reset", + {"type": "reset", "data": {"episode_id": "recorded", "seed": 42}}, + { + "type": "observation", + "data": {"observation": {"counter": 0}, "done": False, "reward": None}, + }, + ), + exchange( + "state", + {"type": "state"}, + {"type": "state", "data": {"episode_id": "recorded", "step_count": 0}}, + ), + exchange( + "step", + {"type": "step", "data": {"increment": 1}}, + { + "type": "observation", + "data": {"observation": {"counter": 1}, "done": False, "reward": 0.5}, + }, + ), + exchange( + "state", + {"type": "state"}, + {"type": "state", "data": {"episode_id": "recorded", "step_count": 1}}, + ), + observation_schema={ + "type": "object", + "properties": {"counter": {"type": "integer"}}, + "required": ["counter"], + }, + ) + + +def rebuild(directory): + metadata = json.loads((directory / "collector-evidence.json").read_text()) + trace = json.loads((directory / metadata["trace_file"]).read_text()) + return RuntimeEvidence( + exchanges=tuple( + WireExchange( + row["operation"], + json.dumps(row["request_json"]), + json.dumps(row["response_json"]), + ) + for row in trace + ), + observation_schema_json=( + json.dumps(metadata["observation_schema"]) + if metadata["schema_available"] + else None + ), + failure_phase=metadata["failure_phase"], + failure_reason=metadata["failure_reason"], + ) + + +def test_saved_collector_evidence_replays_all_basic_graders(tmp_path): + original = measured() + validation_report = report() + write_runtime_bundle(tmp_path, validation_report, evidence=original) + replay = rebuild(tmp_path) + subject = Subject( + tmp_path, validation_report.manifest, None, None, tmp_path, original + ) + for grader in ( + RewardWellFormedGrader(), + ObservationSchemaGrader(), + StateContractGrader(), + ): + assert ( + grader.run(subject).status + == grader.run(replace(subject, runtime_evidence=replay)).status + ) + assert json.loads(replay.observation_schema_json) == json.loads( + original.observation_schema_json + ) + metadata = json.loads((tmp_path / "collector-evidence.json").read_text()) + assert metadata["evidence_schema_version"] == "1" + assert metadata["complete"] is True + assert metadata["redacted"] is False + sums = (tmp_path / "SHA256SUMS").read_text() + assert ( + hashlib.sha256((tmp_path / "collector-evidence.json").read_bytes()).hexdigest() + in sums + ) + + +def test_saved_schema_reproduces_observation_schema_failure(tmp_path): + original = replace( + measured(), observation_schema_json='{"required":["missing-field"]}' + ) + validation_report = report() + write_runtime_bundle(tmp_path, validation_report, evidence=original) + subject = Subject( + tmp_path, validation_report.manifest, None, None, tmp_path, original + ) + original_result = ObservationSchemaGrader().run(subject) + replayed_result = ObservationSchemaGrader().run( + replace(subject, runtime_evidence=rebuild(tmp_path)) + ) + assert original_result.status.value == "fail" + assert original_result.status == replayed_result.status + assert original_result.evidence == replayed_result.evidence + + +def test_truncated_trace_metadata_preserves_the_collection_failure(tmp_path): + original = replace( + measured(), failure_phase="step", failure_reason="step failed (TimeoutError)" + ) + write_runtime_bundle(tmp_path, report(), evidence=original) + metadata = json.loads((tmp_path / "collector-evidence.json").read_text()) + assert metadata["complete"] is False + replayed = rebuild(tmp_path) + assert replayed.failure_phase == original.failure_phase + assert replayed.failure_reason == original.failure_reason + + +def test_schema_and_trace_redaction_marks_evidence_as_modified(tmp_path): + original = replace( + measured(), + observation_schema_json=json.dumps({"description": "hf_notarealtoken12345"}), + failure_reason="Authorization: Bearer not-a-real-secret", + ) + write_runtime_bundle(tmp_path, report(), evidence=original) + metadata = json.loads((tmp_path / "collector-evidence.json").read_text()) + assert metadata["redacted"] is True + assert metadata["observation_schema"]["description"] == "[REDACTED]" + assert "not-a-real-secret" not in (tmp_path / "collector-evidence.json").read_text() + + +def test_deep_subject_json_is_bounded_in_artifacts(tmp_path): + nested = '"leaf"' + for _ in range(600): + nested = '{"child":' + nested + "}" + original = RuntimeEvidence( + exchanges=(WireExchange("step", "{}", nested),), + observation_schema_json=nested, + failure_phase="step", + failure_reason="step failed (ValueError)", + ) + write_runtime_bundle(tmp_path, report(), evidence=original) + metadata = json.loads((tmp_path / "collector-evidence.json").read_text()) + assert metadata["redacted"] is True + assert ( + "artifact nesting limit exceeded" + in (tmp_path / "collector-trace.json").read_text() + ) + assert (tmp_path / "collector-trace.json").stat().st_size < 20_000 + + +def test_nonfinite_wire_values_remain_valid_artifact_json(tmp_path): + original = RuntimeEvidence( + exchanges=(WireExchange("step", "{}", '{"reward":NaN}'),), + observation_schema_json="{}", + ) + write_runtime_bundle(tmp_path, report(), evidence=original) + + def reject_constant(value): + raise ValueError(value) + + trace = json.loads( + (tmp_path / "collector-trace.json").read_text(), parse_constant=reject_constant + ) + assert trace[0]["response_json"]["reward"] == {"invalid_number": "nan"} + + +def test_missing_schema_is_distinct_from_an_advertised_null_schema(tmp_path): + for schema in (None, "null"): + directory = tmp_path / ("missing" if schema is None else "null") + original = replace(measured(), observation_schema_json=schema) + write_runtime_bundle(directory, report(), evidence=original) + assert rebuild(directory).observation_schema_json == schema + + +def test_malformed_wire_omission_is_visible_in_metadata(tmp_path): + original = RuntimeEvidence( + exchanges=(WireExchange("step", "{}", "not JSON"),), + failure_phase="step", + failure_reason="step failed (JSONDecodeError)", + ) + write_runtime_bundle(tmp_path, report(), evidence=original) + metadata = json.loads((tmp_path / "collector-evidence.json").read_text()) + assert metadata["redacted"] is True + assert metadata["omitted_trace_fields"] == [ + {"exchange_index": 0, "field": "response_json"} + ] diff --git a/tests/test_validation/test_runtime_collector.py b/tests/test_validation/test_runtime_collector.py new file mode 100644 index 0000000000..0741e2e8a5 --- /dev/null +++ b/tests/test_validation/test_runtime_collector.py @@ -0,0 +1,107 @@ +"""Hostile HTTP schema responses must stay bounded before JSON/schema grading.""" + +import json + +import httpx +import pytest +from openenv.validation.runtime import collector +from openenv.validation.runtime.contracts import RuntimePlan + + +class TrackedStream(httpx.SyncByteStream): + def __init__(self, payload, *, reject_read=False): + self.payload = payload + self.reject_read = reject_read + self.read = False + self.closed = False + + def __iter__(self): + self.read = True + if self.reject_read: + raise AssertionError("Compressed bytes must never reach the decoder") + yield self.payload + + def close(self): + self.closed = True + + +@pytest.fixture +def plan(): + return RuntimePlan.model_validate( + { + "plan_schema_version": "1", + "reset": {"episode_id": "bounded", "seed": 7}, + "actions": [{"increment": 1}], + } + ) + + +def schema_transport(monkeypatch, stream, headers=None): + requests = [] + original_client = httpx.Client + + def respond(request): + requests.append(request) + return httpx.Response(200, headers=headers or {}, stream=stream) + + monkeypatch.setattr( + collector.httpx, + "Client", + lambda **kwargs: original_client( + transport=httpx.MockTransport(respond), **kwargs + ), + ) + + def no_websocket(*args, **kwargs): + raise ConnectionError("stop after checking schema retrieval") + + monkeypatch.setattr(collector, "connect", no_websocket) + return requests + + +@pytest.mark.parametrize( + "encoding", ["gzip", "deflate", "br", "zstd", "gzip, identity", "unknown"] +) +def test_compressed_schema_is_rejected_before_body_read(monkeypatch, plan, encoding): + stream = TrackedStream(b"untrusted compressed bytes", reject_read=True) + requests = schema_transport(monkeypatch, stream, {"Content-Encoding": encoding}) + evidence = collector.collect_runtime_evidence( + "http://127.0.0.1:8000", plan, episode_timeout_s=2 + ) + assert requests[0].headers["Accept-Encoding"] == "identity" + assert not stream.read + assert stream.closed + assert evidence.failure_phase == "schema" + assert evidence.failure_reason == "schema failed (ValueError)" + assert evidence.observation_schema_json is None + assert evidence.exchanges == () + + +@pytest.mark.parametrize( + "headers", + [{}, {"Content-Encoding": "identity"}, {"Content-Encoding": " Identity "}], +) +def test_uncompressed_schema_is_read_and_retained(monkeypatch, plan, headers): + schema = {"type": "object"} + stream = TrackedStream(json.dumps({"observation": schema}).encode()) + requests = schema_transport(monkeypatch, stream, headers) + evidence = collector.collect_runtime_evidence( + "http://127.0.0.1:8000", plan, episode_timeout_s=2 + ) + assert requests[0].headers["Accept-Encoding"] == "identity" + assert stream.read and stream.closed + assert json.loads(evidence.observation_schema_json) == schema + assert evidence.failure_phase == "connect" + + +def test_uncompressed_schema_still_obeys_total_byte_budget(monkeypatch, plan): + stream = TrackedStream(b"x" * 65) + schema_transport(monkeypatch, stream) + monkeypatch.setattr(collector, "MAX_MESSAGE_BYTES", 64) + evidence = collector.collect_runtime_evidence( + "http://127.0.0.1:8000", plan, episode_timeout_s=2 + ) + assert stream.read and stream.closed + assert evidence.failure_phase == "schema" + assert evidence.failure_reason == "schema failed (ValueError)" + assert evidence.observation_schema_json is None diff --git a/tests/test_validation/test_runtime_execution.py b/tests/test_validation/test_runtime_execution.py new file mode 100644 index 0000000000..dede150e77 --- /dev/null +++ b/tests/test_validation/test_runtime_execution.py @@ -0,0 +1,283 @@ +"""Runtime scheduling, capability preflight and lifetime regression tests.""" + +import json +import shutil +from pathlib import Path +from types import SimpleNamespace + +import pytest +from openenv.validation.policy import load_policy, PolicyError +from openenv.validation.providers import StartupError +from openenv.validation.report import CheckResult +from openenv.validation.runner import run_validation, source_digest +from openenv.validation.runtime.artifacts import write_runtime_bundle +from openenv.validation.runtime.scheduler import execute_graders, order_graders +from openenv.validation.types import CheckStatus, Level, ProviderCapability +from support.runtime import evidence, exchange, FakeRuntimeProvider + + +FIXTURE = Path(__file__).parents[1] / "fixtures/validation/runtime/served_probe" + + +@pytest.fixture +def package(tmp_path): + root = tmp_path / "subject" + shutil.copytree(FIXTURE, root, ignore=shutil.ignore_patterns("__pycache__")) + return root + + +def measured_episode(): + rows = [] + for step in (0, 1): + operation = "reset" if step == 0 else "step" + rows.append( + exchange( + operation, + {"data": {"episode_id": "validation-probe"}}, + { + "type": "observation", + "data": { + "observation": {"counter": step}, + "reward": float(step), + "done": bool(step), + }, + }, + ) + ) + rows.append( + exchange( + "state", + {"type": "state"}, + { + "type": "state", + "data": {"episode_id": "validation-probe", "step_count": step}, + }, + ) + ) + return evidence( + *rows, + observation_schema={ + "type": "object", + "required": ["counter", "reward", "done"], + "properties": { + "counter": {"type": "integer"}, + "done": {"type": "boolean"}, + "reward": {"type": "number"}, + }, + }, + ) + + +def test_runtime_collects_once_cleans_up_and_marks_remaining_work( + package, monkeypatch, tmp_path +): + provider = FakeRuntimeProvider() + calls = [] + + def collect(*args, **kwargs): + calls.append((args, kwargs)) + return measured_episode() + + monkeypatch.setattr("openenv.validation.runner.collect_runtime_evidence", collect) + bundle = tmp_path / "bundle" + report = run_validation( + package, max_level=Level.RUNTIME, provider=provider, artifacts_dir=bundle + ) + results = {r.check_id: r.status for r in report.results} + assert len(calls) == 1 + assert len(provider.builds) == len(provider.launches) == 1 + assert provider.subject.stopped + assert report.report_schema_version == "2" + assert report.levels_run == [Level.STATIC, Level.RUNTIME] + assert report.verdict.value == "warn" + for name in ( + "startup", + "reward_well_formed", + "observation_schema", + "state_contract", + ): + assert results[f"runtime.{name}"] is CheckStatus.PASS + assert results["runtime.network_policy"] is CheckStatus.SKIP + assert json.loads((bundle / "cleanup.json").read_text())["completed"] is True + assert json.loads((bundle / "runtime-plan.json").read_text())["reset"]["seed"] == 42 + + +def test_skip_build_has_no_provider_side_effects(package): + provider = FakeRuntimeProvider() + report = run_validation( + package, max_level=Level.RUNTIME, provider=provider, skip_build=True + ) + assert not provider.builds and not provider.launches + assert report.levels_run == [Level.STATIC] + assert all( + r.status is CheckStatus.SKIP + for r in report.results + if r.check_id.startswith("runtime.") + ) + + +@pytest.mark.parametrize("change", ["network", "gpu", "build"]) +def test_unsupported_capability_is_refused_before_build(package, change): + import yaml + + source = package / "openenv.yaml" + data = yaml.safe_load(source.read_text()) + provider = FakeRuntimeProvider() + if change == "network": + data["validation"]["network"] = {"mode": "no-network"} + elif change == "gpu": + data["validation"]["resources"]["gpus"] = 1 + else: + provider.capabilities = frozenset() + source.write_text(yaml.safe_dump(data)) + report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + assert not provider.builds and not provider.launches + assert ( + next(r for r in report.results if r.check_id == "runtime.startup").status + is CheckStatus.SKIP + ) + + +def test_explicit_v1_rejected_before_runtime(package): + provider = FakeRuntimeProvider() + with pytest.raises(PolicyError, match="policy v2"): + run_validation( + package, + max_level=Level.RUNTIME, + policy=load_policy("v1"), + provider=provider, + ) + assert not provider.builds + + +def test_invalid_plan_is_visible_failure(package): + (package / "validation/runtime.json").write_text('{"actions": []}') + provider = FakeRuntimeProvider() + report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + assert report.verdict.value == "fail" + assert not provider.builds + + +def test_startup_failure_does_not_masquerade_as_skips(package): + provider = FakeRuntimeProvider() + + def failed_build(*args): + raise StartupError("subject build failed") + + provider.build = failed_build + report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + result = next(r for r in report.results if r.check_id == "runtime.startup") + assert result.status is CheckStatus.FAIL + assert report.verdict.value == "fail" + + +@pytest.mark.parametrize("failure", [RuntimeError, KeyboardInterrupt]) +def test_collector_crash_or_cancel_always_tears_down(package, monkeypatch, failure): + provider = FakeRuntimeProvider() + + def explode(*args, **kwargs): + raise failure("token=must-not-appear") + + monkeypatch.setattr("openenv.validation.runner.collect_runtime_evidence", explode) + report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + assert provider.subject.stopped + assert report.verdict.value == "fail" + assert "must-not-appear" not in report.model_dump_json() + + +def test_bad_static_bounds_do_not_suppress_independent_runtime(package, monkeypatch): + path = package / "openenv.yaml" + path.write_text(path.read_text().replace("floor_margin: 0.5", "floor_margin: 0.01")) + provider = FakeRuntimeProvider() + monkeypatch.setattr( + "openenv.validation.runner.collect_runtime_evidence", + lambda *a, **k: measured_episode(), + ) + report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + assert report.results[0].status is CheckStatus.FAIL + assert ( + next(r for r in report.results if r.check_id == "runtime.state_contract").status + is CheckStatus.PASS + ) + + +def test_semantic_ceiling_does_not_claim_semantic_execution(package): + report = run_validation(package, max_level=Level.SEMANTIC, skip_build=True) + assert report.levels_run == [Level.STATIC] + assert any( + r.check_id == "semantic.oracle_max" and r.status is CheckStatus.SKIP + for r in report.results + ) + + +def grader(check_id, depends_on=(), *, status=CheckStatus.PASS, requires=frozenset()): + return SimpleNamespace( + check_id=check_id, + depends_on=depends_on, + requires_provider=requires, + run=lambda _: CheckResult(check_id=check_id, status=status, duration_s=0), + ) + + +def test_scheduler_orders_dependencies_and_keeps_independent_checks(): + first = grader("runtime.z", status=CheckStatus.FAIL) + dependent = grader("runtime.a", ("runtime.z",)) + independent = grader("runtime.other") + assert [g.check_id for g in order_graders([dependent, first])] == [ + "runtime.z", + "runtime.a", + ] + results = { + r.check_id: r for r in execute_graders([dependent, first, independent], None) + } + assert results["runtime.a"].status is CheckStatus.SKIP + assert "runtime.z" in results["runtime.a"].evidence[0] + assert results["runtime.other"].status is CheckStatus.PASS + + +def test_scheduler_detects_cycle_and_missing_prerequisites(): + with pytest.raises(PolicyError, match="cycle"): + order_graders( + [grader("runtime.a", ("runtime.b",)), grader("runtime.b", ("runtime.a",))] + ) + (result,) = execute_graders( + [ + grader( + "runtime.a", + ("runtime.missing",), + requires=frozenset({ProviderCapability.EXEC}), + ) + ], + None, + ) + assert result.status is CheckStatus.SKIP + assert "runtime.missing" in result.evidence[0] + assert "exec" in result.evidence[1] + + +def test_grader_cannot_substitute_a_different_result_id(): + wrong = grader("runtime.wrong") + wrong.check_id = "runtime.expected" + (result,) = execute_graders([wrong], None) + assert result.check_id == "runtime.expected" + assert result.status is CheckStatus.ERROR + + +def test_digest_rejects_symlinks_without_reading_the_target(tmp_path): + root = tmp_path / "subject" + root.mkdir() + (root / "link").symlink_to(tmp_path / "missing-secret") + with pytest.raises(ValueError, match="symbolic"): + source_digest(root) + + +def test_artifacts_encode_invalid_numbers_and_redact_secrets(package, tmp_path): + report = run_validation(package, max_level=Level.RUNTIME, skip_build=True) + raw = evidence(exchange("step", {}, {"token": "secret", "reward": float("nan")})) + write_runtime_bundle(tmp_path / "bundle", report, evidence=raw) + payload = (tmp_path / "bundle/collector-trace.json").read_text() + assert "secret" not in payload + parsed = json.loads( + payload, parse_constant=lambda x: pytest.fail(f"invalid JSON number {x}") + ) + assert parsed[0]["response_json"]["reward"] == {"invalid_number": "nan"} diff --git a/tests/test_validation/test_runtime_grading.py b/tests/test_validation/test_runtime_grading.py new file mode 100644 index 0000000000..96502944c4 --- /dev/null +++ b/tests/test_validation/test_runtime_grading.py @@ -0,0 +1,347 @@ +import copy +import json +import subprocess +import time +from dataclasses import replace + +import pytest +from conftest import load_fixture_manifest +from openenv.validation.graders import Subject +from openenv.validation.graders.runtime import ( + basic, + ObservationSchemaGrader, + RewardWellFormedGrader, + StateContractGrader, +) +from openenv.validation.manifest import NormalizedManifest +from openenv.validation.runtime.contracts import RuntimeEvidence, WireExchange +from openenv.validation.types import CheckStatus +from support.runtime import evidence, exchange + +GRADERS = [RewardWellFormedGrader, ObservationSchemaGrader, StateContractGrader] +OBSERVATION_SCHEMA = { + "type": "object", + "properties": { + "counter": {"type": "integer"}, + "reward": {"type": ["number", "null"]}, + "done": {"type": "boolean"}, + }, + "required": ["counter", "reward", "done"], +} + + +def good_rows(): + return [ + exchange( + "reset", + {"type": "reset", "data": {"episode_id": "measured-episode", "seed": 42}}, + { + "type": "observation", + "data": {"observation": {"counter": 0}, "reward": None, "done": False}, + }, + ), + exchange( + "state", + {"type": "state"}, + { + "type": "state", + "data": {"episode_id": "measured-episode", "step_count": 0}, + }, + ), + exchange( + "step", + {"type": "step", "data": {"increment": 1}}, + { + "type": "observation", + "data": {"observation": {"counter": 1}, "reward": 1.0, "done": False}, + }, + ), + exchange( + "state", + {"type": "state"}, + { + "type": "state", + "data": {"episode_id": "measured-episode", "step_count": 1}, + }, + ), + ] + + +def subject_with(tmp_path, rows=None, schema=OBSERVATION_SCHEMA): + return Subject( + root=tmp_path, + manifest=NormalizedManifest.model_validate( + load_fixture_manifest("served_min_pass") + ), + image_ref=None, + running=None, + outputs_dir=tmp_path, + runtime_evidence=evidence( + *(good_rows() if rows is None else rows), observation_schema=schema + ), + ) + + +def mutate_response(rows, index, mutate): + payload = json.loads(rows[index].response_json) + mutate(payload) + rows[index] = replace(rows[index], response_json=json.dumps(payload)) + return rows + + +@pytest.mark.parametrize("grader", GRADERS) +def test_good_measured_session_passes_each_basic_contract(tmp_path, grader): + assert grader().run(subject_with(tmp_path)).status is CheckStatus.PASS + + +@pytest.mark.parametrize( + "reward", + [ + True, + False, + None, + "0.5", + [], + {}, + float("nan"), + float("inf"), + float("-inf"), + 10**400, + -0.1, + 1.1, + ], +) +def test_step_rewards_are_checked_without_coercion(tmp_path, reward): + rows = mutate_response( + good_rows(), 2, lambda response: response["data"].update(reward=reward) + ) + result = RewardWellFormedGrader().run(subject_with(tmp_path, rows)) + assert result.status is CheckStatus.FAIL + assert any("reward" in message for message in result.evidence) + + +def test_missing_step_reward_is_an_explicit_failure(tmp_path): + rows = mutate_response( + good_rows(), 2, lambda response: response["data"].pop("reward") + ) + result = RewardWellFormedGrader().run(subject_with(tmp_path, rows)) + assert result.status is CheckStatus.FAIL + assert "exchange 2: missing reward" in result.evidence + + +@pytest.mark.parametrize("reward", [0, 1, 0.5]) +def test_numeric_reward_endpoints_and_fraction_are_valid(tmp_path, reward): + rows = mutate_response( + good_rows(), 2, lambda response: response["data"].update(reward=reward) + ) + assert ( + RewardWellFormedGrader().run(subject_with(tmp_path, rows)).status + is CheckStatus.PASS + ) + + +@pytest.mark.parametrize("grader", GRADERS) +def test_reset_only_is_incomplete_not_a_passing_step_contract(tmp_path, grader): + rows = mutate_response( + good_rows()[:2], 0, lambda response: response["data"].update(done=True) + ) + result = grader().run(subject_with(tmp_path, rows)) + assert result.status is CheckStatus.SKIP + assert any("no step" in message for message in result.evidence) + + +@pytest.mark.parametrize("grader", GRADERS) +def test_missing_evidence_is_a_named_skip(tmp_path, grader): + subject = replace(subject_with(tmp_path), runtime_evidence=None) + result = grader().run(subject) + assert result.status is CheckStatus.SKIP + assert result.evidence == ["runtime evidence is unavailable"] + + +@pytest.mark.parametrize("grader", GRADERS) +def test_truncated_collection_cannot_pass_from_a_valid_prefix(tmp_path, grader): + subject = subject_with(tmp_path) + subject = replace( + subject, + runtime_evidence=replace( + subject.runtime_evidence, + failure_phase="step", + failure_reason="step failed (TimeoutError)", + ), + ) + assert grader().run(subject).status is CheckStatus.FAIL + + +@pytest.mark.parametrize( + "schema", [{"type": "not-a-json-type"}, {"required": "counter"}, None] +) +def test_invalid_or_missing_advertised_schema_fails(tmp_path, schema): + result = ObservationSchemaGrader().run(subject_with(tmp_path, schema=schema)) + assert result.status is CheckStatus.FAIL + assert any("schema" in message for message in result.evidence) + + +@pytest.mark.parametrize( + "reference", ["https://example.invalid/schema", "file:///etc/passwd", "other.json"] +) +@pytest.mark.parametrize("keyword", ["$ref", "$dynamicRef"]) +def test_nonlocal_schema_references_are_rejected_before_starting_worker( + tmp_path, monkeypatch, keyword, reference +): + def no_worker(*args, **kwargs): + pytest.fail("untrusted external references must not reach schema evaluation") + + monkeypatch.setattr(basic.subprocess, "run", no_worker) + result = ObservationSchemaGrader().run( + subject_with(tmp_path, schema={keyword: reference}) + ) + assert result.status is CheckStatus.FAIL + assert result.evidence == ["observation schema has a non-local reference"] + + +def test_internal_schema_references_are_supported(tmp_path): + schema = copy.deepcopy(OBSERVATION_SCHEMA) + schema["$defs"] = {"counter": {"type": "integer"}} + schema["properties"]["counter"] = {"$ref": "#/$defs/counter"} + assert ( + ObservationSchemaGrader().run(subject_with(tmp_path, schema=schema)).status + is CheckStatus.PASS + ) + + +def test_schema_reconstructs_reward_and_done_from_envelope(tmp_path): + # These are deliberately absent from the nested observation, as on the wire. + rows = good_rows() + assert "reward" not in json.loads(rows[2].response_json)["data"]["observation"] + assert ( + ObservationSchemaGrader().run(subject_with(tmp_path, rows)).status + is CheckStatus.PASS + ) + + +@pytest.mark.parametrize( + "field,value", [("done", None), ("done", "false"), ("done", 0), ("observation", [])] +) +def test_invalid_raw_envelope_is_rejected_before_defaults(tmp_path, field, value): + rows = mutate_response( + good_rows(), 2, lambda response: response["data"].update({field: value}) + ) + result = ObservationSchemaGrader().run(subject_with(tmp_path, rows)) + assert result.status is CheckStatus.FAIL + assert "exchange 2: malformed observation envelope" in result.evidence + + +def test_missing_done_is_not_filled_by_a_model_default(tmp_path): + rows = mutate_response( + good_rows(), 2, lambda response: response["data"].pop("done") + ) + result = ObservationSchemaGrader().run(subject_with(tmp_path, rows)) + assert result.status is CheckStatus.FAIL + assert "exchange 2: malformed observation envelope" in result.evidence + + +def test_schema_mismatch_evidence_does_not_echo_private_subject_values(tmp_path): + private_value = "not-a-number-secret-observation" + rows = mutate_response( + good_rows(), + 2, + lambda response: response["data"]["observation"].update(counter=private_value), + ) + result = ObservationSchemaGrader().run(subject_with(tmp_path, rows)) + assert result.status is CheckStatus.FAIL + assert any("counter" in message for message in result.evidence) + assert private_value not in result.model_dump_json() + + +def test_schema_worker_deadline_becomes_a_validation_finding(tmp_path, monkeypatch): + def timeout(command, **kwargs): + assert command[-1].endswith("schema_worker.py") + assert 0 < kwargs["timeout"] <= 5 + raise subprocess.TimeoutExpired(command, kwargs["timeout"]) + + monkeypatch.setattr(basic.subprocess, "run", timeout) + result = ObservationSchemaGrader().run(subject_with(tmp_path)) + assert result.status is CheckStatus.FAIL + assert result.evidence == ["observation schema evaluation exceeded its time budget"] + + +def test_pathological_regex_runs_in_killable_worker(tmp_path, monkeypatch): + run = subprocess.run + + def short_deadline(command, **kwargs): + kwargs["timeout"] = 0.5 + return run(command, **kwargs) + + monkeypatch.setattr(basic.subprocess, "run", short_deadline) + schema = copy.deepcopy(OBSERVATION_SCHEMA) + schema["properties"]["counter"] = {"type": "string", "pattern": "^(a+)+$"} + rows = good_rows() + for index in (0, 2): + mutate_response( + rows, + index, + lambda response: response["data"]["observation"].update( + counter="a" * 100 + "!" + ), + ) + started = time.monotonic() + result = ObservationSchemaGrader().run(subject_with(tmp_path, rows, schema=schema)) + assert time.monotonic() - started < 3 + assert result.status is CheckStatus.FAIL + assert any("time budget" in message for message in result.evidence) + + +@pytest.mark.parametrize("episode_id", ["other-episode", None, 42]) +def test_state_identity_must_match_the_requested_episode(tmp_path, episode_id): + rows = mutate_response( + good_rows(), 3, lambda response: response["data"].update(episode_id=episode_id) + ) + result = StateContractGrader().run(subject_with(tmp_path, rows)) + assert result.status is CheckStatus.FAIL + assert "exchange 3: episode_id differs from reset" in result.evidence + + +@pytest.mark.parametrize("count", [True, 1.0, "1", None, -1, 0, 2]) +def test_state_count_is_strict_and_matches_successful_steps(tmp_path, count): + rows = mutate_response( + good_rows(), 3, lambda response: response["data"].update(step_count=count) + ) + result = StateContractGrader().run(subject_with(tmp_path, rows)) + assert result.status is CheckStatus.FAIL + assert "exchange 3: incorrect step_count" in result.evidence + + +def test_reset_state_count_must_start_at_zero(tmp_path): + rows = mutate_response( + good_rows(), 1, lambda response: response["data"].update(step_count=5) + ) + assert ( + StateContractGrader().run(subject_with(tmp_path, rows)).status + is CheckStatus.FAIL + ) + + +def test_missing_state_snapshots_do_not_pass(tmp_path): + rows = [row for row in good_rows() if row.operation != "state"] + result = StateContractGrader().run(subject_with(tmp_path, rows)) + assert result.status is CheckStatus.FAIL + assert "no state snapshots were measured" in result.evidence + + +@pytest.mark.parametrize("grader", GRADERS) +def test_malformed_wire_is_a_finding_not_a_validator_crash(tmp_path, grader): + rows = good_rows() + rows[2] = WireExchange("step", rows[2].request_json, "{not-json") + # Collector failure is part of the evidence even for graders that do not read + # step response bodies (the state check only needs the step request count). + subject = subject_with(tmp_path, rows) + subject = replace( + subject, + runtime_evidence=RuntimeEvidence( + exchanges=subject.runtime_evidence.exchanges, + observation_schema_json=subject.runtime_evidence.observation_schema_json, + failure_phase="step", + failure_reason="step failed (JSONDecodeError)", + ), + ) + assert grader().run(subject).status is CheckStatus.FAIL diff --git a/tests/validation_runtime/uv.lock b/tests/validation_runtime/uv.lock index f261bed87f..40460643a2 100644 --- a/tests/validation_runtime/uv.lock +++ b/tests/validation_runtime/uv.lock @@ -921,6 +921,7 @@ dependencies = [ { name = "gradio" }, { name = "httpx" }, { name = "huggingface-hub" }, + { name = "jsonschema" }, { name = "openai" }, { name = "packaging" }, { name = "pydantic" }, @@ -947,6 +948,7 @@ requires-dist = [ { name = "httpx", specifier = ">=0.28.1" }, { name = "huggingface-hub", specifier = ">=0.20.0" }, { name = "inspect-ai", marker = "extra == 'inspect'", specifier = ">=0.3.0" }, + { name = "jsonschema", specifier = ">=4.20.0" }, { name = "modal", marker = "extra == 'modal'", specifier = ">=1.4.0" }, { name = "novita-sandbox", marker = "extra == 'novita'", specifier = ">=2.1.1" }, { name = "openai", specifier = ">=2.7.2" }, From a74c97bce42efa9afeea422bb0c800c2f8961408 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Mon, 21 Sep 2026 14:39:12 +0200 Subject: [PATCH 02/29] Exercise runtime faults and reference Echo with complete evidence --- scripts/validation/reproduce.py | 59 ++++++ scripts/validation/verify_artifacts.py | 28 ++- .../validation/graders/runtime/basic.py | 6 +- src/openenv/validation/runner.py | 6 +- src/openenv/validation/runtime/collector.py | 24 ++- tests/fixtures/validation/runtime/README.md | 7 + .../validation/runtime/echo_canary/README.md | 18 ++ .../runtime/echo_canary/execution.json | 7 + .../echo_canary/validation/runtime.json | 16 ++ .../validation/runtime/served_probe/app.py | 23 ++- .../integration/test_echo_canary.py | 112 +++++++++++ .../integration/test_runtime_cli.py | 183 ++++++++++++++++-- tests/test_validation/test_reproduction.py | 56 +++++- tests/test_validation/test_runtime_unicode.py | 57 ++++++ tests/validation_runtime/README.md | 25 ++- tests/validation_runtime/acceptance.json | 28 +++ 16 files changed, 628 insertions(+), 27 deletions(-) create mode 100644 tests/fixtures/validation/runtime/echo_canary/README.md create mode 100644 tests/fixtures/validation/runtime/echo_canary/execution.json create mode 100644 tests/fixtures/validation/runtime/echo_canary/validation/runtime.json create mode 100644 tests/test_validation/integration/test_echo_canary.py create mode 100644 tests/test_validation/test_runtime_unicode.py create mode 100644 tests/validation_runtime/acceptance.json diff --git a/scripts/validation/reproduce.py b/scripts/validation/reproduce.py index 7872a1c14f..c6586353b8 100644 --- a/scripts/validation/reproduce.py +++ b/scripts/validation/reproduce.py @@ -20,6 +20,7 @@ ROOT = Path(__file__).resolve().parents[2] PROJECT = ROOT / "tests/validation_runtime" FIXTURE = ROOT / "tests/fixtures/validation/runtime/served_probe" +ECHO_OVERLAY = ROOT / "tests/fixtures/validation/runtime/echo_canary" MAX_LOG_BYTES = 1024 * 1024 @@ -349,6 +350,56 @@ def stage_image(work, output, pins, manifest): return context, image +def stage_echo_canary(work, context, output, manifest): + """Stage unchanged reference-environment sources without importing them.""" + import yaml + + source = ROOT / "envs/echo_env" + if any(path.is_symlink() for path in source.rglob("*")): + raise RuntimeError("Echo source snapshots do not permit symlinks") + target = work / "echo-subject" + target.mkdir() + shutil.copytree( + source, + target / "echo_env", + ignore=shutil.ignore_patterns("__pycache__", "*.pyc", ".venv", "*.egg-info"), + ) + copied = hashes(target / "echo_env") + if any(digest(source / name) != value for name, value in copied.items()): + raise RuntimeError("Echo source changed while the canary was staged") + # The wheelhouse and pinned, offline installation are shared with the probe. + shutil.copytree(context / "wheelhouse", target / "wheelhouse") + shutil.copyfile(context / "requirements.txt", target / "requirements.txt") + dockerfile = (context / "Dockerfile").read_text() + substitutions = { + "COPY served_probe /app/served_probe": "COPY echo_env /app/echo_env", + '["python", "-m", "served_probe.app"]': '["python", "-m", "echo_env.server.app"]', + } + for before, after in substitutions.items(): + if dockerfile.count(before) != 1: + raise RuntimeError("Shared image recipe changed; review the Echo canary") + dockerfile = dockerfile.replace(before, after) + (target / "Dockerfile").write_text(dockerfile) + declaration = yaml.safe_load((source / "openenv.yaml").read_text()) + declaration["validation"]["execution"] = json.loads( + (ECHO_OVERLAY / "execution.json").read_text() + ) + (target / "openenv.yaml").write_text(yaml.safe_dump(declaration, sort_keys=False)) + shutil.copytree(ECHO_OVERLAY / "validation", target / "validation") + provenance = { + "source_directory": "envs/echo_env", + "source_hashes": copied, + "overlay_hashes": hashes(ECHO_OVERLAY), + "dockerfile_sha256": digest(target / "Dockerfile"), + "manifest_sha256": digest(target / "openenv.yaml"), + "wheel_sha256": manifest["wheel_sha256"], + "subject_imported_on_host": False, + } + manifest["echo_canary"] = provenance + write_json(output / "echo-canary-source.json", provenance) + return target + + def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument( @@ -368,11 +419,14 @@ def main(): run_id = f"l2-{uuid.uuid4().hex[:12]}" output = (args.output / run_id).resolve() output.mkdir(parents=True) + inventory = PROJECT / "acceptance.json" + shutil.copyfile(inventory, output / "acceptance.json") print(f"Evidence: {output}", flush=True) manifest = { "schema_version": "1", "run_id": run_id, "suite": args.suite, + "acceptance_inventory_sha256": digest(inventory), "argv": sys.argv, "head_sha": run(["git", "rev-parse", "HEAD"]), "dirty": bool(run(["git", "status", "--porcelain"])), @@ -389,6 +443,7 @@ def main(): status = 1 environment = os.environ.copy() environment.pop("PYTHONPATH", None) + environment.pop("PYTEST_ADDOPTS", None) environment["OPENENV_VALIDATION_ARTIFACTS"] = str(output) environment["OPENENV_REQUIRE_COMPLETE"] = "1" if args.require_complete else "0" try: @@ -417,10 +472,14 @@ def main(): ) } context, image = stage_image(Path(temporary), output, pins, manifest) + echo_context = stage_echo_canary( + Path(temporary), context, output, manifest + ) environment.update( { "OPENENV_VALIDATION_IMAGE": image, "OPENENV_VALIDATION_CONTEXT": str(context), + "OPENENV_VALIDATION_ECHO_CONTEXT": str(echo_context), "OPENENV_REQUIRE_DOCKER": "1", } ) diff --git a/scripts/validation/verify_artifacts.py b/scripts/validation/verify_artifacts.py index 7d1d48a947..06745f3b4f 100644 --- a/scripts/validation/verify_artifacts.py +++ b/scripts/validation/verify_artifacts.py @@ -7,6 +7,10 @@ from pathlib import Path from xml.etree import ElementTree +ACCEPTANCE = ( + Path(__file__).resolve().parents[2] / "tests/validation_runtime/acceptance.json" +) + def verify(root): root = root.resolve() @@ -29,14 +33,34 @@ def verify(root): manifest = json.loads((root / "run-manifest.json").read_text()) if not manifest["success"]: raise ValueError("Run did not complete successfully") - suites = ElementTree.parse(root / "junit.xml").getroot().iter("testsuite") + inventory_bytes = (root / "acceptance.json").read_bytes() + if inventory_bytes != ACCEPTANCE.read_bytes(): + raise ValueError("Run does not match the committed acceptance inventory") + if hashlib.sha256(inventory_bytes).hexdigest() != manifest.get( + "acceptance_inventory_sha256" + ): + raise ValueError("Acceptance inventory digest differs from run manifest") + inventory = json.loads(inventory_bytes)["suites"] + selected_suite = manifest.get("suite") + if selected_suite not in {"fast", *inventory}: + raise ValueError("Unknown acceptance suite in run manifest") + document = ElementTree.parse(root / "junit.xml").getroot() count = 0 - for suite in suites: + for suite in document.iter("testsuite"): count += int(suite.get("tests", "0")) if any(int(suite.get(key, "0")) for key in ("failures", "errors", "skipped")): raise ValueError("Required test suite contains failures, errors or skips") if not count: raise ValueError("Empty test suite cannot establish completion") + observed = { + f"{case.get('classname', '')}::{case.get('name', '')}" + for case in document.iter("testcase") + } + missing = set(inventory.get(selected_suite, [])) - observed + if missing: + raise ValueError( + "Missing required acceptance cases: " + ", ".join(sorted(missing)) + ) return count diff --git a/src/openenv/validation/graders/runtime/basic.py b/src/openenv/validation/graders/runtime/basic.py index f76960005b..4df69a9368 100644 --- a/src/openenv/validation/graders/runtime/basic.py +++ b/src/openenv/validation/graders/runtime/basic.py @@ -143,7 +143,11 @@ def check(self, subject, evidence) -> list[str]: try: checked = subprocess.run( [sys.executable, "-I", str(worker)], - input=json.dumps({"schema": schema, "observations": observations}), + input=json.dumps( + {"schema": schema, "observations": observations}, + ensure_ascii=False, + separators=(",", ":"), + ), capture_output=True, text=True, timeout=5, diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index f4d24a177d..016e4260be 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -22,7 +22,7 @@ from .providers import ProviderError, StartupError, UnsupportedCapability from .report import CheckResult, ValidationReport, ValidationReportV2 from .runtime.artifacts import write_runtime_bundle -from .runtime.collector import collect_runtime_evidence +from .runtime.collector import collect_runtime_evidence, RuntimeCollectionInterrupted from .runtime.contracts import LaunchSpec, load_runtime_plan, RuntimePlanError from .runtime.scheduler import execute_graders from .signature import detect_signature @@ -209,7 +209,9 @@ def _runtime(subject, *, skip_build, provider): started=started, ) checks = [] - except KeyboardInterrupt: + except KeyboardInterrupt as exc: + if isinstance(exc, RuntimeCollectionInterrupted): + evidence = exc.evidence result = _outcome( "runtime.startup", CheckStatus.ERROR, diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index f3cf318a98..cb4c613309 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -13,6 +13,14 @@ MAX_TRACE_BYTES = 8 * 1024 * 1024 +class RuntimeCollectionInterrupted(KeyboardInterrupt): + """Cancellation carrying the completed, immutable episode prefix.""" + + def __init__(self, evidence: RuntimeEvidence): + super().__init__("runtime collection interrupted") + self.evidence = evidence + + def collect_runtime_evidence( base_url: str, plan: RuntimePlan, @@ -78,7 +86,12 @@ def remaining() -> float: schema = json.loads(payload) if not isinstance(schema, dict) or "observation" not in schema: raise ValueError("missing observation schema") - schema_json = json.dumps(schema["observation"], allow_nan=False) + schema_json = json.dumps( + schema["observation"], + allow_nan=False, + ensure_ascii=False, + separators=(",", ":"), + ) endpoint = urlsplit(base_url) ws_url = urlunsplit( @@ -148,6 +161,15 @@ def exchange(operation: str, data: dict | None = None) -> dict: return RuntimeEvidence( exchanges=tuple(exchanges), observation_schema_json=schema_json ) + except KeyboardInterrupt: + raise RuntimeCollectionInterrupted( + RuntimeEvidence( + exchanges=tuple(exchanges), + observation_schema_json=schema_json, + failure_phase=phase, + failure_reason=f"{phase} failed (KeyboardInterrupt)", + ) + ) from None except Exception as exc: # Exception text may include submitted payloads or URL credentials. return RuntimeEvidence( diff --git a/tests/fixtures/validation/runtime/README.md b/tests/fixtures/validation/runtime/README.md index e70c052feb..be3274b875 100644 --- a/tests/fixtures/validation/runtime/README.md +++ b/tests/fixtures/validation/runtime/README.md @@ -11,6 +11,13 @@ work, not evidence that a grader exists. During the first three slices, the exercise startup and the basic runtime contracts. `good` means those four checks pass; it does not mean all Level 2 checks have been implemented. +The first-slice regression inventory additionally covers a missing `done` field, +a blocked second step, controlled CLI interruption and the unchanged reference +Echo environment. `tests/validation_runtime/acceptance.json` records the required +protocol and Docker test identities; a filtered subset cannot establish suite +completion. The Echo canary deliberately records its reward/schema compatibility +findings instead of modifying the environment to make it pass. + The `served_probe/` subject is shared by real-protocol, Docker and installed-wheel tests. Its fault modes are test-only and each introduces one intentional defect. The public `validation/runtime.json` file contains only reset inputs and actions. diff --git a/tests/fixtures/validation/runtime/echo_canary/README.md b/tests/fixtures/validation/runtime/echo_canary/README.md new file mode 100644 index 0000000000..cf243fd961 --- /dev/null +++ b/tests/fixtures/validation/runtime/echo_canary/README.md @@ -0,0 +1,18 @@ +# Reference Echo environment canary + +The lab copies `envs/echo_env` from the tested checkout byte for byte and records +its source hashes. This directory supplies only the runtime execution declaration +and public action plan. It does not contain a substitute Echo implementation. + +The canary reuses the shared probe's digest-pinned base, exact OpenEnv wheel, +hash-checked wheelhouse and offline installation. Its image recipe changes only +the subject COPY and entrypoint. The validator never imports Echo on the host. + +The acceptance test requires real tool responses and a continuing episode with +state counts 0, 1 and 2. The current Echo implementation emits null step rewards +and resets with a base observation despite advertising CallToolObservation as its +schema. Consequently the expected validator result is FAIL with explicit reward +and schema findings, while startup and state checks pass. A passing acceptance +test proves those compatibility findings are observed; it does not certify Echo +or complete Level 2 validation. When the environment contract changes, update the +expected findings together with evidence of the intended behavior. diff --git a/tests/fixtures/validation/runtime/echo_canary/execution.json b/tests/fixtures/validation/runtime/echo_canary/execution.json new file mode 100644 index 0000000000..1661edd9e9 --- /dev/null +++ b/tests/fixtures/validation/runtime/echo_canary/execution.json @@ -0,0 +1,7 @@ +{ + "kind": "openenv_ws", + "probe_path": "validation/runtime.json", + "dockerfile": "Dockerfile", + "context": ".", + "agent_boundary": "api" +} diff --git a/tests/fixtures/validation/runtime/echo_canary/validation/runtime.json b/tests/fixtures/validation/runtime/echo_canary/validation/runtime.json new file mode 100644 index 0000000000..40115d072e --- /dev/null +++ b/tests/fixtures/validation/runtime/echo_canary/validation/runtime.json @@ -0,0 +1,16 @@ +{ + "plan_schema_version": "1", + "reset": {"episode_id": "echo-canary", "seed": 42, "options": {}}, + "actions": [ + { + "type": "call_tool", + "tool_name": "echo_message", + "arguments": {"message": "echo canary first"} + }, + { + "type": "call_tool", + "tool_name": "echo_with_length", + "arguments": {"message": "echo canary second"} + } + ] +} diff --git a/tests/fixtures/validation/runtime/served_probe/app.py b/tests/fixtures/validation/runtime/served_probe/app.py index 39feaf4899..91999cb91c 100644 --- a/tests/fixtures/validation/runtime/served_probe/app.py +++ b/tests/fixtures/validation/runtime/served_probe/app.py @@ -1,5 +1,6 @@ """A deterministic subject, with deliberate faults confined to test assets.""" +import asyncio import json import os from typing import Any @@ -54,6 +55,25 @@ def __init__(self, application, mode): self.mode = mode async def __call__(self, scope, receive, send): + steps = 0 + + async def fault_receive(): + nonlocal steps + message = await receive() + if ( + self.mode == "hung_step" + and message["type"] == "websocket.receive" + and message.get("text") + and json.loads(message["text"]).get("type") == "step" + ): + steps += 1 + if steps == 2: + # Tests signal the CLI only after the collector has completed + # a real reset, first step and both corresponding state reads. + os.write(1, b"OPENENV_VALIDATION_STEP_BLOCKED\n") + await asyncio.Event().wait() + return message + async def fault_send(message: dict[str, Any]): if message["type"] == "websocket.send" and message.get("text"): data = json.loads(message["text"]) @@ -70,7 +90,7 @@ async def fault_send(message: dict[str, Any]): message = {**message, "text": json.dumps(data)} await send(message) - await self.application(scope, receive, fault_send) + await self.application(scope, fault_receive, fault_send) def make_app(mode="good"): @@ -84,6 +104,7 @@ def make_app(mode="good"): "bad_observation", "missing_done", "bad_state", + "hung_step", }: raise ValueError(f"Unknown fixture mode: {mode}") return WireFault( diff --git a/tests/test_validation/integration/test_echo_canary.py b/tests/test_validation/integration/test_echo_canary.py new file mode 100644 index 0000000000..c3c3859f6a --- /dev/null +++ b/tests/test_validation/integration/test_echo_canary.py @@ -0,0 +1,112 @@ +"""Run the unchanged reference Echo environment through the installed CLI.""" + +import hashlib +import json +import os +from pathlib import Path + +import pytest +from test_runtime_cli import _invoke_cli + + +pytestmark = pytest.mark.docker + + +def test_reference_echo_replays_real_tools_and_exposes_contract_findings(tmp_path): + configured = os.environ.get("OPENENV_VALIDATION_ECHO_CONTEXT") + if not configured: + if os.environ.get("OPENENV_REQUIRE_DOCKER") == "1": + pytest.fail("Required Echo canary needs OPENENV_VALIDATION_ECHO_CONTEXT") + pytest.skip("Run the validation lab to stage the exact-source Echo canary") + context = Path(configured) + evidence_root = Path(os.environ["OPENENV_VALIDATION_ARTIFACTS"]) + provenance = json.loads((evidence_root / "echo-canary-source.json").read_text()) + assert provenance["source_directory"] == "envs/echo_env" + assert provenance["subject_imported_on_host"] is False + assert provenance["source_hashes"] + for name, expected in provenance["source_hashes"].items(): + assert ( + hashlib.sha256((context / "echo_env" / name).read_bytes()).hexdigest() + == expected + ) + + result, report, checks, artifacts = _invoke_cli(context, tmp_path, "echo_canary") + expected = { + "runtime.startup": "pass", + "runtime.state_contract": "pass", + "runtime.reward_well_formed": "fail", + "runtime.observation_schema": "fail", + } + (evidence_root / "cli/echo_canary/compatibility-findings.json").write_text( + json.dumps( + { + "source_directory": "envs/echo_env", + "source_digest": report["source_digest"], + "image_ref": checks["runtime.startup"]["measured"].get("image_ref"), + "expected": expected, + "observed": { + check_id: { + "status": checks[check_id]["status"], + "evidence": checks[check_id]["evidence"], + } + for check_id in expected + }, + "interpretation": "Expected contract findings; Echo is not certified.", + }, + indent=2, + sort_keys=True, + ) + + "\n" + ) + assert report["levels_run"] == [1, 2] + assert report["manifest"]["name"] == "echo_env" + assert checks["static.manifest"]["status"] == "pass" + assert checks["runtime.startup"]["status"] == "pass" + assert checks["runtime.state_contract"]["status"] == "pass" + assert artifacts["cleanup.json"]["required"] is True + trace = artifacts["collector-trace.json"] + assert [row["operation"] for row in trace] == [ + "reset", + "state", + "step", + "state", + "step", + "state", + ] + states = [ + row["response_json"]["data"] for row in trace if row["operation"] == "state" + ] + assert [state["episode_id"] for state in states] == ["echo-canary"] * 3 + assert [state["step_count"] for state in states] == [0, 1, 2] + steps = [ + row["response_json"]["data"] for row in trace if row["operation"] == "step" + ] + for step, tool_name, message in zip( + steps, + ("echo_message", "echo_with_length"), + ("echo canary first", "echo canary second"), + strict=True, + ): + observation = step["observation"] + assert observation["tool_name"] == tool_name + assert observation["error"] is None + assert message in json.dumps(observation["result"]) + assert step["reward"] is None + assert step["done"] is False + + # Preserve genuine compatibility findings instead of modifying Echo or + # accepting a permissive validator result merely to turn this canary green. + assert result.returncode == 1 + assert report["verdict"] == "fail" + assert checks["runtime.reward_well_formed"]["status"] == "fail" + assert checks["runtime.observation_schema"]["status"] == "fail" + assert any( + "finite and in range" in entry + for entry in checks["runtime.reward_well_formed"]["evidence"] + ) + assert checks["runtime.observation_schema"]["evidence"] + assert "tool_name" not in trace[0]["response_json"]["data"]["observation"] + metadata = artifacts["collector-evidence.json"] + assert metadata["complete"] is True + assert metadata["failure_phase"] is None + assert "tool_name" in metadata["observation_schema"]["required"] diff --git a/tests/test_validation/integration/test_runtime_cli.py b/tests/test_validation/integration/test_runtime_cli.py index 630e9716a4..daee42a28e 100644 --- a/tests/test_validation/integration/test_runtime_cli.py +++ b/tests/test_validation/integration/test_runtime_cli.py @@ -4,8 +4,10 @@ import json import os import shutil +import signal import subprocess import sys +import time from importlib.resources import files from pathlib import Path @@ -66,7 +68,23 @@ def cli_context(tmp_path): return context -def _container_ids(): +def _case_label(context): + return hashlib.sha256(str(context).encode()).hexdigest()[:20] + + +def _mark_context(context): + # An image label is inherited by every container, including failed starts. + # This scopes external cleanup verification to this exact test invocation. + dockerfile = context / "Dockerfile" + content = dockerfile.read_bytes() + dockerfile.unlink() + dockerfile.write_bytes( + content + + f"\nLABEL org.openenv.validation.test={_case_label(context)}\n".encode() + ) + + +def _container_ids(context): checked = subprocess.run( [ "docker", @@ -75,7 +93,7 @@ def _container_ids(): "--all", "--quiet", "--filter", - "label=org.openenv.validation.run", + f"label=org.openenv.validation.test={_case_label(context)}", ], capture_output=True, text=True, @@ -85,6 +103,36 @@ def _container_ids(): return set(checked.stdout.split()) +def _wait_for_blocked_step(process, context, work): + deadline = time.monotonic() + 120 + while time.monotonic() < deadline: + if process.poll() is not None: + pytest.fail("CLI exited before the deliberate blocked step") + for container_id in _container_ids(context): + logs = subprocess.run( + ["docker", "logs", "--tail", "30", container_id], + capture_output=True, + text=True, + timeout=5, + ) + if "OPENENV_VALIDATION_STEP_BLOCKED" in logs.stdout: + (work / "blocked-handshake.json").write_text( + json.dumps( + { + "container_id": container_id, + "marker": "OPENENV_VALIDATION_STEP_BLOCKED", + "milestone": "second_step_received", + } + ) + + "\n" + ) + return time.monotonic() + # Poll an observable protocol milestone rather than guessing when the + # CLI has reached its collector after an arbitrarily long image build. + time.sleep(0.05) + pytest.fail("Subject never reached its deliberately blocked second step") + + def _verify_bundle(bundle, report): schema = json.loads( files("openenv.validation") @@ -111,10 +159,19 @@ def _verify_bundle(bundle, report): return {name: json.loads((bundle / name).read_text()) for name in recorded} -def _invoke_cli(context, tmp_path, case, *, skip_build=False): +def _invoke_cli( + context, + tmp_path, + case, + *, + skip_build=False, + wait_for_blocked=False, + interrupt=False, +): artifact_root = Path(os.environ.get("OPENENV_VALIDATION_ARTIFACTS", tmp_path)) work = artifact_root / "cli" / case work.mkdir(parents=True, exist_ok=True) + _mark_context(context) report_path = work / "report.json" environment = os.environ.copy() environment.pop("PYTHONPATH", None) @@ -147,24 +204,66 @@ def _invoke_cli(context, tmp_path, case, *, skip_build=False): ] if skip_build: command.append("--skip-build") - before = _container_ids() + process = None try: - result = subprocess.run( + with ( + (work / "stdout.json").open("w") as stdout, + (work / "stderr.log").open("w") as stderr, + ): + process = subprocess.Popen( + command, + cwd=work, + env=environment, + stdout=stdout, + stderr=stderr, + start_new_session=True, + ) + blocked_at = None + if wait_for_blocked: + blocked_at = _wait_for_blocked_step(process, context, work) + if interrupt: + assert blocked_at is not None, "Signal requires a protocol handshake" + process.send_signal(signal.SIGINT) + process.wait(timeout=15 if blocked_at else 180) + if blocked_at is not None: + elapsed = time.monotonic() - blocked_at + (work / "termination.json").write_text( + json.dumps( + {"seconds_after_blocked_step": elapsed, "signal": interrupt} + ) + + "\n" + ) + assert elapsed < 15, "CLI did not terminate within its time budget" + result = subprocess.CompletedProcess( command, - cwd=work, - env=environment, - capture_output=True, - text=True, - timeout=180, + process.returncode, + (work / "stdout.json").read_text(), + (work / "stderr.log").read_text(), ) finally: - remaining = _container_ids() - before + if process is not None and process.poll() is None: + os.killpg(process.pid, signal.SIGKILL) + process.wait(timeout=10) + remaining = _container_ids(context) (work / "cleanup-external.json").write_text( - json.dumps({"remaining_container_ids": sorted(remaining)}) + "\n" + json.dumps( + { + "test_label": _case_label(context), + "remaining_container_ids": sorted(remaining), + } + ) + + "\n" ) + if remaining: + # Retain the failed assertion and evidence, but do not leave a failed + # acceptance test consuming resources or touching other test runs. + subprocess.run( + ["docker", "rm", "--force", "--volumes", *sorted(remaining)], + capture_output=True, + timeout=15, + check=True, + ) assert not remaining, "CLI validation leaked a container" - (work / "stdout.json").write_text(result.stdout) - (work / "stderr.log").write_text(result.stderr) assert report_path.is_file(), result.stderr report = json.loads(report_path.read_text()) assert json.loads(result.stdout) == report @@ -192,6 +291,7 @@ def _invoke_cli(context, tmp_path, case, *, skip_build=False): ("good", None), ("bad_reward", "runtime.reward_well_formed"), ("bad_observation", "runtime.observation_schema"), + ("missing_done", "runtime.observation_schema"), ("bad_state", "runtime.state_contract"), ], ) @@ -229,6 +329,61 @@ def test_cli_runtime_contract_findings(cli_context, tmp_path, mode, failed_check assert set(artifacts["coverage.json"]["incomplete"]) == PENDING +def _assert_partial_episode(artifacts, expected_failure): + trace = artifacts["collector-trace.json"] + assert [row["operation"] for row in trace] == ["reset", "state", "step", "state"] + assert trace[0]["response_json"]["data"]["observation"]["counter"] == 0 + assert trace[2]["response_json"]["data"]["observation"]["counter"] == 1 + assert trace[3]["response_json"]["data"]["step_count"] == 1 + evidence = artifacts["collector-evidence.json"] + assert evidence["complete"] is False + assert evidence["failure_phase"] == "step" + assert expected_failure in evidence["failure_reason"] + assert artifacts["cleanup.json"] == {"required": True, "completed": True} + + +def test_cli_hung_step_times_out_with_partial_evidence(cli_context, tmp_path): + with (cli_context / "Dockerfile").open("a") as stream: + stream.write("\nENV VALIDATION_FAULT=hung_step\n") + manifest = cli_context / "openenv.yaml" + content = manifest.read_text() + assert content.count("episode_timeout_s: 30.0") == 1 + content = content.replace("episode_timeout_s: 30.0", "episode_timeout_s: 3.0") + # The context hardlinks immutable inputs; changing one must not alter the + # shared source used by subsequent test cases. + manifest.unlink() + manifest.write_text(content) + result, report, checks, artifacts = _invoke_cli( + cli_context, tmp_path, "hung_step", wait_for_blocked=True + ) + assert result.returncode == 1 + assert report["verdict"] == "fail" + assert report["manifest"]["resources"]["episode_timeout_s"] == 3.0 + assert checks["runtime.startup"]["status"] == "pass" + assert all( + checks[key]["status"] == "fail" for key in IMPLEMENTED - {"runtime.startup"} + ) + _assert_partial_episode(artifacts, "TimeoutError") + + +def test_cli_sigint_retains_partial_evidence_and_cleans_up(cli_context, tmp_path): + with (cli_context / "Dockerfile").open("a") as stream: + stream.write("\nENV VALIDATION_FAULT=hung_step\n") + result, report, checks, artifacts = _invoke_cli( + cli_context, tmp_path, "sigint", wait_for_blocked=True, interrupt=True + ) + # Recorded check errors fail closed through the existing policy (exit 1). + # Exit 3 is reserved for an internal error that cannot produce this report. + assert result.returncode == 1 + assert report["verdict"] == "fail" + assert checks["runtime.startup"]["status"] == "error" + assert "interrupt" in " ".join(checks["runtime.startup"]["evidence"]).lower() + assert all( + checks[key]["status"] == "skip" for key in IMPLEMENTED - {"runtime.startup"} + ) + _assert_partial_episode(artifacts, "KeyboardInterrupt") + + def test_cli_startup_failure_is_a_finding_and_leaves_no_container( cli_context, tmp_path ): diff --git a/tests/test_validation/test_reproduction.py b/tests/test_validation/test_reproduction.py index 16f5a86335..d4d65b6b4b 100644 --- a/tests/test_validation/test_reproduction.py +++ b/tests/test_validation/test_reproduction.py @@ -7,11 +7,13 @@ import sys import zipfile from pathlib import Path +from xml.etree import ElementTree import pytest ROOT = Path(__file__).parents[2] +ACCEPTANCE = ROOT / "tests/validation_runtime/acceptance.json" def module(name): @@ -97,12 +99,28 @@ def fake_docker(argv, **kwargs): assert (wheelhouse / wheel.name).read_bytes() == wheel.read_bytes() -def evidence(tmp_path, skipped=0): +def evidence(tmp_path, skipped=0, suite="fast", cases=None, inventory=None): reproduction = module("reproduce") - (tmp_path / "run-manifest.json").write_text(json.dumps({"success": True})) - (tmp_path / "junit.xml").write_text( - f'' + inventory = ACCEPTANCE.read_bytes() if inventory is None else inventory + (tmp_path / "acceptance.json").write_bytes(inventory) + (tmp_path / "run-manifest.json").write_text( + json.dumps( + { + "success": True, + "suite": suite, + "acceptance_inventory_sha256": hashlib.sha256(inventory).hexdigest(), + } + ) + ) + cases = ["tests.example::test_pass"] if cases is None else cases + document = ElementTree.Element("testsuites") + test_suite = ElementTree.SubElement( + document, "testsuite", tests=str(len(cases)), skipped=str(skipped) ) + for case in cases: + classname, name = case.split("::", 1) + ElementTree.SubElement(test_suite, "testcase", classname=classname, name=name) + ElementTree.ElementTree(document).write(tmp_path / "junit.xml") entries = reproduction.hashes(tmp_path) (tmp_path / "SHA256SUMS").write_text( "".join(f"{value} {name}\n" for name, value in entries.items()) @@ -128,6 +146,36 @@ def test_required_acceptance_cannot_silently_skip(tmp_path): module("verify_artifacts").verify(tmp_path) +@pytest.mark.parametrize("suite", ["protocol", "docker"]) +def test_required_acceptance_rejects_deselected_cases_despite_passing_tests( + tmp_path, suite +): + required = json.loads(ACCEPTANCE.read_text())["suites"][suite] + evidence(tmp_path, suite=suite, cases=required[:-1]) + with pytest.raises(ValueError, match="Missing required acceptance cases"): + module("verify_artifacts").verify(tmp_path) + + +@pytest.mark.parametrize("suite", ["protocol", "docker"]) +def test_all_required_acceptance_cases_establish_completion(tmp_path, suite): + required = json.loads(ACCEPTANCE.read_text())["suites"][suite] + evidence(tmp_path, suite=suite, cases=required) + assert module("verify_artifacts").verify(tmp_path) == len(required) + + +def test_reduced_inventory_cannot_redefine_required_acceptance(tmp_path): + changed = json.loads(ACCEPTANCE.read_text()) + changed["suites"]["docker"] = changed["suites"]["docker"][:1] + evidence( + tmp_path, + suite="docker", + cases=changed["suites"]["docker"], + inventory=json.dumps(changed).encode(), + ) + with pytest.raises(ValueError, match="committed acceptance inventory"): + module("verify_artifacts").verify(tmp_path) + + def test_nested_case_checksum_is_itself_covered_by_bundle(tmp_path): case = tmp_path / "case" case.mkdir() diff --git a/tests/test_validation/test_runtime_unicode.py b/tests/test_validation/test_runtime_unicode.py new file mode 100644 index 0000000000..4c6d9ee3ef --- /dev/null +++ b/tests/test_validation/test_runtime_unicode.py @@ -0,0 +1,57 @@ +"""Bounded Unicode evidence must survive the schema-worker transport.""" + +import json + +from conftest import load_fixture_manifest +from openenv.validation.graders import Subject +from openenv.validation.graders.runtime import ObservationSchemaGrader +from openenv.validation.manifest import NormalizedManifest +from openenv.validation.runtime.collector import MAX_MESSAGE_BYTES, MAX_TRACE_BYTES +from openenv.validation.runtime.contracts import RuntimeEvidence, WireExchange +from openenv.validation.types import CheckStatus + + +def test_valid_unicode_episode_within_wire_budgets_passes_schema_worker(tmp_path): + exchanges = [] + for index in range(7): + response = json.dumps( + { + "type": "observation", + "data": { + "observation": {"text": "é" * 300_000}, + "reward": 1, + "done": False, + }, + }, + ensure_ascii=False, + ) + exchanges.append( + WireExchange( + operation="reset" if index == 0 else "step", + request_json="{}", + response_json=response, + ) + ) + + frame_sizes = [len(row.response_json.encode()) for row in exchanges] + assert max(frame_sizes) < MAX_MESSAGE_BYTES + assert sum(frame_sizes) < MAX_TRACE_BYTES + subject = Subject( + root=tmp_path, + manifest=NormalizedManifest.model_validate( + load_fixture_manifest("served_min_pass") + ), + image_ref=None, + running=None, + outputs_dir=tmp_path, + runtime_evidence=RuntimeEvidence( + exchanges=tuple(exchanges), + observation_schema_json=json.dumps( + {"type": "object", "properties": {"text": {"type": "string"}}} + ), + ), + ) + + result = ObservationSchemaGrader().run(subject) + + assert result.status is CheckStatus.PASS, result.evidence diff --git a/tests/validation_runtime/README.md b/tests/validation_runtime/README.md index eb08d68d8b..75ddafb443 100644 --- a/tests/validation_runtime/README.md +++ b/tests/validation_runtime/README.md @@ -27,10 +27,27 @@ checkout with `PYTHONPATH` removed, exercising installed package data and the production OpenEnv `/ws` endpoint. Each launch uses a fresh subject. The image supports controlled `VALIDATION_FAULT` modes: `good`, `bad_reward`, -`bad_observation`, `missing_done`, `bad_state`, and `startup_failure`. All fault +`bad_observation`, `missing_done`, `bad_state`, `hung_step`, and `startup_failure`. All fault switches and wire corruption remain inside test assets. They share one fixture and one public runtime plan, so a defect changes one property at a time. +The Docker suite contains 13 required cases: three provider lifecycle tests, +nine CLI fault/control cases, and one real `echo_env` canary. The hung-step case +checks the episode deadline; the interruption case sends SIGINT only after a +container log confirms the second step has begun. Both must retain the completed +reset/state/step/state prefix and remove their own containers. Each CLI case uses +a unique image label for independent cleanup verification. + +The Echo canary copies the actual `envs/echo_env` sources unchanged and records +their hashes. A test overlay adds only the execution declaration, replay plan and +pinned offline image recipe. It runs `echo_message` and `echo_with_length` in one +session and verifies episode identity and state counts 0, 1, 2. Echo currently +returns null step rewards and a reset observation that lacks the advertised +`tool_name` field: the canary therefore expects explicit reward/schema **FAIL** +findings and CLI exit 1. Its passing test means those compatibility findings were +observed correctly; it does not mean Echo passed runtime validation. Inspect +`cli/echo_canary/compatibility-findings.json` for the actual results. + Evidence is written to `outputs/validation-runtime//`, including source, fixture, lock and wheel hashes; the retained wheelhouse; platform and toolchain details; test results and bounded logs; provider cleanup evidence; and checksums. @@ -43,7 +60,11 @@ python scripts/validation/verify_artifacts.py outputs/validation-runtime/ Date: Mon, 21 Sep 2026 14:55:45 +0200 Subject: [PATCH 03/29] fix: bound runtime transport and worker inputs --- .../validation/graders/runtime/basic.py | 14 +- src/openenv/validation/runtime/collector.py | 76 ++++++- .../validation/runtime/schema_worker.py | 18 +- .../test_validation/test_runtime_transport.py | 213 ++++++++++++++++++ tests/test_validation/test_runtime_unicode.py | 50 +++- 5 files changed, 351 insertions(+), 20 deletions(-) create mode 100644 tests/test_validation/test_runtime_transport.py diff --git a/src/openenv/validation/graders/runtime/basic.py b/src/openenv/validation/graders/runtime/basic.py index 4df69a9368..7937bdf38a 100644 --- a/src/openenv/validation/graders/runtime/basic.py +++ b/src/openenv/validation/graders/runtime/basic.py @@ -132,9 +132,12 @@ def check(self, subject, evidence) -> list[str]: ): problems.append(f"exchange {index}: malformed observation envelope") continue - observation = dict(data["observation"]) - observation.update(reward=data["reward"], done=data["done"]) - observations.append({"index": index, "observation": observation}) + # Preserve the bounded wire representation across the subprocess + # boundary: normalizing numbers such as 1e9 can inflate a valid + # episode beyond the worker's input limit. + observations.append( + {"index": index, "response_json": exchange.response_json} + ) if count == 0: problems.append("no observations were measured") # A subject-supplied regex or recursive schema can exhaust CPU. Keep all @@ -144,7 +147,10 @@ def check(self, subject, evidence) -> list[str]: checked = subprocess.run( [sys.executable, "-I", str(worker)], input=json.dumps( - {"schema": schema, "observations": observations}, + { + "schema_json": evidence.observation_schema_json, + "observations": observations, + }, ensure_ascii=False, separators=(",", ":"), ), diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index cb4c613309..1160dbbcd8 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -1,6 +1,8 @@ """Bounded raw protocol collection in one OpenEnv orchestration session.""" import json +import socket +import threading import time from urllib.parse import urlsplit, urlunsplit @@ -13,6 +15,49 @@ MAX_TRACE_BYTES = 8 * 1024 * 1024 +def _abort_transport(connection): + # The send thread may hold websockets' protocol lock. Shut down the raw + # transport directly so sendall and its concurrent receiver can both exit. + try: + connection.socket.shutdown(socket.SHUT_RDWR) + except OSError: + pass # The peer or another cleanup path may already have closed it. + try: + connection.socket.close() + except OSError: + pass # A concurrent close must not replace the original operation error. + + +def _bounded_call(connection, operation, timeout_s): + if timeout_s <= 0: + raise TimeoutError("episode deadline exceeded") + expired = threading.Event() + deadline = time.monotonic() + timeout_s + + def abort(): + expired.set() + _abort_transport(connection) + + watchdog = threading.Timer(timeout_s, abort) + watchdog.daemon = True + watchdog.start() + try: + operation() + except KeyboardInterrupt: + _abort_transport(connection) + raise + except Exception: + if expired.is_set(): + raise TimeoutError("transport deadline exceeded") from None + raise + finally: + watchdog.cancel() + watchdog.join() + if expired.is_set() or time.monotonic() >= deadline: + _abort_transport(connection) + raise TimeoutError("transport deadline exceeded") + + class RuntimeCollectionInterrupted(KeyboardInterrupt): """Cancellation carrying the completed, immutable episode prefix.""" @@ -104,7 +149,7 @@ def remaining() -> float: ) ) phase = "connect" - with connect( + connection = connect( ws_url, proxy=None, open_timeout=remaining(), @@ -112,7 +157,9 @@ def remaining() -> float: max_size=MAX_MESSAGE_BYTES, max_queue=1, compression=None, - ) as socket: + ) + complete = False + try: def exchange(operation: str, data: dict | None = None) -> dict: nonlocal phase, trace_bytes @@ -121,9 +168,10 @@ def exchange(operation: str, data: dict | None = None) -> dict: if data is not None: request["data"] = data request_json = json.dumps(request, allow_nan=False) - remaining() - socket.send(request_json) - raw = socket.recv(timeout=remaining()) + _bounded_call( + connection, lambda: connection.send(request_json), remaining() + ) + raw = connection.recv(timeout=remaining()) if not isinstance(raw, str): raise ValueError("binary response is not the JSON protocol") trace_bytes += len(raw.encode("utf-8")) + len( @@ -157,7 +205,23 @@ def exchange(operation: str, data: dict | None = None) -> dict: break observation = exchange("step", action) exchange("state") - socket.send(json.dumps({"type": "close"})) + complete = True + finally: + # Teardown is best effort and cannot replace an in-episode failure + # or invalidate an otherwise completely measured episode. + if complete: + try: + _bounded_call( + connection, + lambda: connection.send(json.dumps({"type": "close"})), + min(remaining(), 1.0), + ) + except Exception: + pass # A server may close immediately after its final response. + try: + _bounded_call(connection, connection.close, 1.0) + except Exception: + _abort_transport(connection) return RuntimeEvidence( exchanges=tuple(exchanges), observation_schema_json=schema_json ) diff --git a/src/openenv/validation/runtime/schema_worker.py b/src/openenv/validation/runtime/schema_worker.py index 90251ce2fe..169645613f 100644 --- a/src/openenv/validation/runtime/schema_worker.py +++ b/src/openenv/validation/runtime/schema_worker.py @@ -6,6 +6,11 @@ from jsonschema import Draft202012Validator from referencing import Registry +# Quoting the 8 MiB raw trace costs at most 16 MiB. The remainder covers the +# 1 MiB schema after conservative 6x JSON normalization and 2x quoting, plus row +# metadata. Limits apply to bytes, independently of the text stream's encoding. +MAX_INPUT_BYTES = 32 * 1024 * 1024 + def main(): """Evaluate bounded inputs, printing only error locations, never subject values.""" @@ -17,16 +22,23 @@ def main(): resource.setrlimit(resource.RLIMIT_AS, (512 * 1024 * 1024,) * 2) except (ImportError, ValueError, OSError): pass # The parent always enforces the independent wall-clock deadline. - payload = json.loads(sys.stdin.read(10 * 1024 * 1024)) + raw = sys.stdin.buffer.read(MAX_INPUT_BYTES + 1) + if len(raw) > MAX_INPUT_BYTES: + sys.stdout.write(json.dumps(["schema worker input exceeds its size budget"])) + return + payload = json.loads(raw) problems = [] try: - schema = payload["schema"] + schema = json.loads(payload["schema_json"]) Draft202012Validator.check_schema(schema) # An empty registry has no retrieval callback: unresolved references # cannot trigger host filesystem or network access. validator = Draft202012Validator(schema, registry=Registry()) for row in payload["observations"]: - for error in validator.iter_errors(row["observation"]): + data = json.loads(row["response_json"])["data"] + observation = dict(data["observation"]) + observation.update(reward=data["reward"], done=data["done"]) + for error in validator.iter_errors(observation): location = "/".join(str(x) for x in error.absolute_path)[:160] problems.append( f"exchange {row['index']}: schema mismatch at {location or '/'}" diff --git a/tests/test_validation/test_runtime_transport.py b/tests/test_validation/test_runtime_transport.py new file mode 100644 index 0000000000..b26b3d2393 --- /dev/null +++ b/tests/test_validation/test_runtime_transport.py @@ -0,0 +1,213 @@ +"""Episode findings survive teardown noise and blocked transport writes.""" + +import json +import socket +import threading +import time + +import httpx +import pytest +from openenv.validation.runtime import collector +from openenv.validation.runtime.contracts import RuntimePlan + + +class AbortableTransport: + def __init__(self): + self.aborted = threading.Event() + + def shutdown(self, how): + self.aborted.set() + + def close(self): + self.aborted.set() + + +class EpisodeConnection: + def __init__(self, *, close_send=False, close_failure=False, block_send=False): + self.socket = AbortableTransport() + self.close_send = close_send + self.close_failure = close_failure + self.block_send = block_send + self.step_failure = False + self.closed = False + self.operation = None + self.steps = 0 + + def __enter__(self): + return self + + def __exit__(self, *_args): + self.close() + + def send(self, payload): + self.operation = json.loads(payload)["type"] + if self.block_send: + # The fallback keeps the RED test finite when no watchdog exists. + self.socket.aborted.wait(0.8) + raise ConnectionError("transport did not accept the write") + if self.operation == "close" and self.close_send: + raise ConnectionError("server already closed a complete episode") + + def recv(self, timeout): + if self.operation == "state": + return json.dumps( + { + "type": "state", + "data": {"episode_id": "bounded", "step_count": self.steps}, + } + ) + if self.operation == "step": + if self.step_failure: + raise ValueError("genuine in-episode failure") + self.steps += 1 + return json.dumps( + { + "type": "observation", + "data": { + "observation": {"counter": self.steps}, + "reward": 1, + "done": False, + }, + } + ) + + def close(self): + self.closed = True + if self.close_failure: + raise ConnectionError("websocket teardown failed") + + +@pytest.fixture +def collect(monkeypatch): + original_client = httpx.Client + monkeypatch.setattr( + collector.httpx, + "Client", + lambda **kwargs: original_client( + transport=httpx.MockTransport( + lambda request: httpx.Response( + 200, json={"observation": {"type": "object"}} + ) + ), + **kwargs, + ), + ) + plan = RuntimePlan.model_validate( + { + "plan_schema_version": "1", + "reset": {"episode_id": "bounded", "seed": 7}, + "actions": [{"increment": 1}], + } + ) + + def run(connection, *, episode_timeout_s=2, request_timeout_s=1): + monkeypatch.setattr(collector, "connect", lambda *args, **kwargs: connection) + return collector.collect_runtime_evidence( + "http://127.0.0.1:8000", + plan, + episode_timeout_s=episode_timeout_s, + request_timeout_s=request_timeout_s, + ) + + return run + + +@pytest.mark.parametrize("fault", ["close_send", "close_failure"]) +def test_teardown_noise_does_not_fail_a_completed_episode(collect, fault): + connection = EpisodeConnection(**{fault: True}) + evidence = collect(connection) + assert [row.operation for row in evidence.exchanges] == [ + "reset", + "state", + "step", + "state", + ] + assert evidence.failure_reason is None + assert evidence.failure_phase is None + assert connection.closed + + +def test_teardown_failure_does_not_replace_a_genuine_operation_failure(collect): + connection = EpisodeConnection(close_failure=True) + connection.step_failure = True + evidence = collect(connection) + assert evidence.failure_phase == "step" + assert evidence.failure_reason == "step failed (ValueError)" + assert [row.operation for row in evidence.exchanges] == ["reset", "state"] + assert connection.closed + + +def test_interruption_preserves_partial_evidence_despite_teardown_failure(collect): + connection = EpisodeConnection(close_failure=True) + receive = connection.recv + + def interrupt_step(timeout): + if connection.operation == "step": + raise KeyboardInterrupt + return receive(timeout) + + connection.recv = interrupt_step + with pytest.raises(collector.RuntimeCollectionInterrupted) as interrupted: + collect(connection) + evidence = interrupted.value.evidence + assert [row.operation for row in evidence.exchanges] == ["reset", "state"] + assert evidence.failure_phase == "step" + assert evidence.failure_reason == "step failed (KeyboardInterrupt)" + assert connection.closed + + +@pytest.mark.parametrize( + "timeouts", + [ + {"episode_timeout_s": 0.05, "request_timeout_s": 2}, + {"episode_timeout_s": 2, "request_timeout_s": 0.05}, + ], +) +def test_blocked_send_obeys_episode_and_operation_deadlines(collect, timeouts): + connection = EpisodeConnection(block_send=True) + started = time.monotonic() + evidence = collect(connection, **timeouts) + elapsed = time.monotonic() - started + assert elapsed < 0.5, "send exceeded the declared transport deadline" + assert connection.socket.aborted.is_set() + assert connection.closed + assert evidence.exchanges == () + assert evidence.failure_phase == "reset" + assert evidence.failure_reason == "reset failed (TimeoutError)" + + +def test_blocked_os_send_is_interrupted_and_transport_is_closed(collect): + sender, receiver = socket.socketpair() + sender.setsockopt(socket.SOL_SOCKET, socket.SO_SNDBUF, 4096) + sender.setblocking(False) + try: + while True: + sender.send(b"x" * 4096) + except BlockingIOError: + pass + sender.setblocking(True) + connection = EpisodeConnection() + connection.socket = sender + connection.send = lambda payload: sender.sendall(payload.encode()) + rescued = threading.Event() + + def rescue(): + rescued.set() + sender.shutdown(socket.SHUT_RDWR) + + fallback = threading.Timer(2, rescue) + fallback.daemon = True + fallback.start() + try: + started = time.monotonic() + evidence = collect(connection, episode_timeout_s=0.05) + assert time.monotonic() - started < 1 + assert not rescued.is_set(), "collector needed the external rescue timeout" + assert sender.fileno() == -1 + assert connection.closed + assert evidence.failure_reason == "reset failed (TimeoutError)" + finally: + fallback.cancel() + fallback.join() + sender.close() + receiver.close() diff --git a/tests/test_validation/test_runtime_unicode.py b/tests/test_validation/test_runtime_unicode.py index 4c6d9ee3ef..1242c7bbc1 100644 --- a/tests/test_validation/test_runtime_unicode.py +++ b/tests/test_validation/test_runtime_unicode.py @@ -1,19 +1,24 @@ -"""Bounded Unicode evidence must survive the schema-worker transport.""" +"""Bounded wire evidence must survive the schema-worker transport.""" import json +import subprocess +import sys +from pathlib import Path +import pytest from conftest import load_fixture_manifest from openenv.validation.graders import Subject from openenv.validation.graders.runtime import ObservationSchemaGrader from openenv.validation.manifest import NormalizedManifest from openenv.validation.runtime.collector import MAX_MESSAGE_BYTES, MAX_TRACE_BYTES from openenv.validation.runtime.contracts import RuntimeEvidence, WireExchange +from openenv.validation.runtime.schema_worker import MAX_INPUT_BYTES from openenv.validation.types import CheckStatus -def test_valid_unicode_episode_within_wire_budgets_passes_schema_worker(tmp_path): - exchanges = [] - for index in range(7): +@pytest.mark.parametrize("payload_kind", ["unicode", "numeric"]) +def test_valid_episode_within_wire_budgets_passes_schema_worker(tmp_path, payload_kind): + if payload_kind == "unicode": response = json.dumps( { "type": "observation", @@ -25,6 +30,18 @@ def test_valid_unicode_episode_within_wire_budgets_passes_schema_worker(tmp_path }, ensure_ascii=False, ) + schema = {"type": "object", "properties": {"text": {"type": "string"}}} + else: + # Compact exponent notation is legal JSON. Parsing then serializing this + # episode expands it past the old worker cap despite fitting wire limits. + response = ( + '{"type":"observation","data":{"observation":{"values":[' + + ",".join(["1e9"] * 125_000) + + ']},"reward":1,"done":false}}' + ) + schema = {"type": "object", "properties": {"values": {"type": "array"}}} + exchanges = [] + for index in range(7): exchanges.append( WireExchange( operation="reset" if index == 0 else "step", @@ -46,12 +63,31 @@ def test_valid_unicode_episode_within_wire_budgets_passes_schema_worker(tmp_path outputs_dir=tmp_path, runtime_evidence=RuntimeEvidence( exchanges=tuple(exchanges), - observation_schema_json=json.dumps( - {"type": "object", "properties": {"text": {"type": "string"}}} - ), + observation_schema_json=json.dumps(schema), ), ) result = ObservationSchemaGrader().run(subject) assert result.status is CheckStatus.PASS, result.evidence + + +def test_schema_worker_rejects_oversized_binary_input(): + worker = ( + Path(__file__).parents[2] / "src/openenv/validation/runtime/schema_worker.py" + ) + payload = json.dumps({"schema_json": "{}", "observations": []}).encode() + # Valid JSON plus whitespace demonstrates a byte-bound rejection rather than + # a parse error or a crash after unbounded allocation. + payload += b" " * (MAX_INPUT_BYTES + 1 - len(payload)) + + result = subprocess.run( + [sys.executable, "-I", str(worker)], + input=payload, + capture_output=True, + timeout=5, + env={}, + check=True, + ) + + assert json.loads(result.stdout) == ["schema worker input exceeds its size budget"] From 728248286f4e5254722be1b71e5792b9c0944642 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Mon, 21 Sep 2026 13:07:55 +0000 Subject: [PATCH 04/29] fix: honor runtime timeout and UTF-8 schema input --- src/openenv/validation/graders/runtime/basic.py | 1 + src/openenv/validation/runner.py | 2 +- tests/test_validation/test_runtime_execution.py | 17 +++++++++++++++++ tests/test_validation/test_runtime_grading.py | 1 + 4 files changed, 20 insertions(+), 1 deletion(-) diff --git a/src/openenv/validation/graders/runtime/basic.py b/src/openenv/validation/graders/runtime/basic.py index 7937bdf38a..80247e3124 100644 --- a/src/openenv/validation/graders/runtime/basic.py +++ b/src/openenv/validation/graders/runtime/basic.py @@ -156,6 +156,7 @@ def check(self, subject, evidence) -> list[str]: ), capture_output=True, text=True, + encoding="utf-8", timeout=5, env={}, ) diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index 016e4260be..ffae8529f6 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -157,7 +157,7 @@ def _runtime(subject, *, skip_build, provider): evidence = collect_runtime_evidence( running.base_url, plan, - episode_timeout_s=min(manifest.resources.episode_timeout_s, 300.0), + episode_timeout_s=manifest.resources.episode_timeout_s, ) # A health endpoint without a functioning protocol isn't a startup success. if not evidence.exchanges and evidence.failure_reason: diff --git a/tests/test_validation/test_runtime_execution.py b/tests/test_validation/test_runtime_execution.py index dede150e77..6e48c3f7c1 100644 --- a/tests/test_validation/test_runtime_execution.py +++ b/tests/test_validation/test_runtime_execution.py @@ -102,6 +102,23 @@ def collect(*args, **kwargs): assert json.loads((bundle / "runtime-plan.json").read_text())["reset"]["seed"] == 42 +def test_runtime_collection_uses_declared_episode_timeout(package, monkeypatch): + path = package / "openenv.yaml" + path.write_text( + path.read_text().replace("episode_timeout_s: 30.0", "episode_timeout_s: 600.0") + ) + calls = [] + + def collect(*args, **kwargs): + calls.append(kwargs) + return measured_episode() + + monkeypatch.setattr("openenv.validation.runner.collect_runtime_evidence", collect) + run_validation(package, max_level=Level.RUNTIME, provider=FakeRuntimeProvider()) + + assert calls[0]["episode_timeout_s"] == 600.0 + + def test_skip_build_has_no_provider_side_effects(package): provider = FakeRuntimeProvider() report = run_validation( diff --git a/tests/test_validation/test_runtime_grading.py b/tests/test_validation/test_runtime_grading.py index 96502944c4..d8705f953b 100644 --- a/tests/test_validation/test_runtime_grading.py +++ b/tests/test_validation/test_runtime_grading.py @@ -256,6 +256,7 @@ def test_schema_mismatch_evidence_does_not_echo_private_subject_values(tmp_path) def test_schema_worker_deadline_becomes_a_validation_finding(tmp_path, monkeypatch): def timeout(command, **kwargs): assert command[-1].endswith("schema_worker.py") + assert kwargs["encoding"] == "utf-8" assert 0 < kwargs["timeout"] <= 5 raise subprocess.TimeoutExpired(command, kwargs["timeout"]) From b9ae0debad8793832de8caf674fdc034a6a22be4 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Mon, 21 Sep 2026 19:44:13 +0200 Subject: [PATCH 05/29] fix: address runtime validation reviews --- src/openenv/cli/commands/validate.py | 2 +- .../validation/graders/runtime/basic.py | 4 ++ src/openenv/validation/runner.py | 29 ++++++-- src/openenv/validation/runtime/contracts.py | 41 ++++++++++- .../test_validation/test_runtime_contracts.py | 70 +++++++++++++++++++ .../test_validation/test_runtime_execution.py | 40 ++++++++++- tests/test_validation/test_runtime_grading.py | 22 ++++++ 7 files changed, 196 insertions(+), 12 deletions(-) diff --git a/src/openenv/cli/commands/validate.py b/src/openenv/cli/commands/validate.py index 2d943a123d..55286002f8 100644 --- a/src/openenv/cli/commands/validate.py +++ b/src/openenv/cli/commands/validate.py @@ -154,7 +154,7 @@ def validate( applies the severity policy, and emits a report. Exit codes: 0 pass/warn · 1 fail · 2 unrecognized/unsupported package · 3 - internal error. + internal or policy error. Examples: diff --git a/src/openenv/validation/graders/runtime/basic.py b/src/openenv/validation/graders/runtime/basic.py index 80247e3124..34c71d1098 100644 --- a/src/openenv/validation/graders/runtime/basic.py +++ b/src/openenv/validation/graders/runtime/basic.py @@ -126,6 +126,7 @@ def check(self, subject, evidence) -> list[str]: data = response["data"] if ( response.get("type") != "observation" + or not isinstance(data, dict) or not isinstance(data.get("observation"), dict) or type(data.get("done")) is not bool or "reward" not in data @@ -191,6 +192,9 @@ def check(self, subject, evidence) -> list[str]: states += 1 response = json.loads(exchange.response_json) data = response["data"] + if not isinstance(data, dict): + problems.append(f"exchange {index}: malformed state envelope") + continue if ( response.get("type") != "state" or data.get("episode_id") != episode_id diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index ffae8529f6..6631e44e78 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -185,27 +185,27 @@ def _runtime(subject, *, skip_build, provider): "runtime.startup", CheckStatus.SKIP, str(exc), started=started ) checks = [] - except RuntimePlanError: + except RuntimePlanError as exc: result = _outcome( "runtime.startup", CheckStatus.FAIL, - "runtime plan is missing, unsafe or invalid", + str(exc)[:4096], started=started, ) checks = [] - except StartupError: + except StartupError as exc: result = _outcome( "runtime.startup", CheckStatus.FAIL, - "subject failed build or readiness; inspect the Docker fixture/build inputs", + str(exc)[:4096], started=started, ) checks = [] - except ProviderError: + except ProviderError as exc: result = _outcome( "runtime.startup", CheckStatus.ERROR, - "provider could not complete a bounded operation", + str(exc)[:4096], started=started, ) checks = [] @@ -353,7 +353,22 @@ def run_validation( reason = "unmet dependency: runtime.startup" results.append(_outcome(entry.check_id, CheckStatus.SKIP, reason)) if source_digest(target) != digest_before: - results = [r for r in results if r.check_id != "runtime.startup"] + results = [ + _outcome( + r.check_id, + CheckStatus.SKIP, + "unmet dependency: runtime.startup (package source changed)", + ) + if r.check_id + in { + "runtime.reward_well_formed", + "runtime.observation_schema", + "runtime.state_contract", + } + else r + for r in results + if r.check_id != "runtime.startup" + ] results.append( _outcome( "runtime.startup", diff --git a/src/openenv/validation/runtime/contracts.py b/src/openenv/validation/runtime/contracts.py index dd32e42ad2..3d84e7be68 100644 --- a/src/openenv/validation/runtime/contracts.py +++ b/src/openenv/validation/runtime/contracts.py @@ -7,7 +7,14 @@ from pathlib import Path from typing import Any, Literal -from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator +from pydantic import ( + BaseModel, + ConfigDict, + Field, + field_validator, + model_validator, + ValidationError, +) from ..manifest import ExecutionDeclaration, NetworkPolicy, ResourceDeclaration @@ -113,7 +120,7 @@ def _unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]: result = {} for key, value in pairs: if key in result: - raise ValueError(f"duplicate runtime JSON key: {key}") + raise ValueError("duplicate runtime JSON key") result[key] = value return result @@ -147,7 +154,35 @@ def load_runtime_plan(root: Path, execution: ExecutionDeclaration) -> RuntimePla raw = json.loads(payload, object_pairs_hook=_unique_object) _check_json(raw) return RuntimePlan.model_validate(raw) - except (OSError, ValueError, RecursionError) as exc: + except ValidationError as exc: + fields = RuntimePlan.model_fields.keys() | RuntimeReset.model_fields.keys() + errors = [] + for error in exc.errors( + include_input=False, include_context=False, include_url=False + )[:5]: + location = ".".join( + str(part) if isinstance(part, int) or part in fields else "" + for part in error["loc"] + ) + # These messages come from the fixed schema and validators; raw inputs, + # exception context and subject-controlled field names are omitted. + errors.append(f"{location or ''}: {error['type']} ({error['msg']})") + if exc.error_count() > 5: + errors.append(f"... ({exc.error_count()} errors total)") + raise RuntimePlanError( + "runtime plan schema validation failed: " + "; ".join(errors) + ) from exc + except json.JSONDecodeError as exc: + raise RuntimePlanError( + f"invalid runtime plan JSON at line {exc.lineno}, column {exc.colno}: {exc.msg}" + ) from exc + except UnicodeError as exc: + raise RuntimePlanError("runtime plan contains invalid text encoding") from exc + except OSError as exc: + raise RuntimePlanError( + f"runtime plan could not be read ({type(exc).__name__})" + ) from exc + except (ValueError, RecursionError) as exc: raise RuntimePlanError(f"invalid runtime plan: {exc}") from exc diff --git a/tests/test_validation/test_runtime_contracts.py b/tests/test_validation/test_runtime_contracts.py index 295838affc..15541b0cba 100644 --- a/tests/test_validation/test_runtime_contracts.py +++ b/tests/test_validation/test_runtime_contracts.py @@ -95,6 +95,76 @@ def test_duplicate_json_fields_are_not_ambiguous(tmp_path): load_runtime_plan(tmp_path, ExecutionDeclaration()) +def test_plan_schema_errors_omit_inputs_and_unknown_field_names(tmp_path): + data = plan_data() + data["reset"]["seed"] = "private-input-value" + data["private-extra-field"] = "private-extra-value" + write_plan(tmp_path, json.dumps(data)) + + with pytest.raises(RuntimePlanError) as error: + load_runtime_plan(tmp_path, ExecutionDeclaration()) + + diagnostic = str(error.value) + assert "reset.seed: int_type" in diagnostic + assert ": extra_forbidden" in diagnostic + assert "private-" not in diagnostic + + +def test_plan_schema_error_diagnostics_are_bounded(tmp_path): + data = plan_data() + data["actions"] = ["private-input-value"] * 100 + write_plan(tmp_path, json.dumps(data)) + + with pytest.raises(RuntimePlanError) as error: + load_runtime_plan(tmp_path, ExecutionDeclaration()) + + diagnostic = str(error.value) + assert diagnostic.count("dict_type") == 5 + assert "100 errors total" in diagnostic + assert "private-input-value" not in diagnostic + assert len(diagnostic) < 1024 + + +def test_plan_schema_error_preserves_safe_validator_reason(tmp_path): + data = plan_data() + data["reset"]["options"] = {"seed": "private-input-value"} + write_plan(tmp_path, json.dumps(data)) + with pytest.raises(RuntimePlanError) as error: + load_runtime_plan(tmp_path, ExecutionDeclaration()) + assert "reset options cannot override episode_id or seed" in str(error.value) + assert "private-input-value" not in str(error.value) + + +def test_duplicate_json_diagnostic_omits_key(tmp_path): + write_plan(tmp_path, '{"private-duplicate-key": 1, "private-duplicate-key": 2}') + with pytest.raises(RuntimePlanError, match="duplicate runtime JSON key") as error: + load_runtime_plan(tmp_path, ExecutionDeclaration()) + assert "private-duplicate-key" not in str(error.value) + + +def test_missing_plan_diagnostic_omits_filesystem_paths(tmp_path): + with pytest.raises(RuntimePlanError) as error: + load_runtime_plan(tmp_path, ExecutionDeclaration()) + assert str(error.value) == "runtime plan could not be read (FileNotFoundError)" + + +def test_invalid_plan_json_reports_location_without_input(tmp_path): + write_plan(tmp_path, '{"private-key": private-value}') + with pytest.raises(RuntimePlanError) as error: + load_runtime_plan(tmp_path, ExecutionDeclaration()) + assert str(error.value) == ( + "invalid runtime plan JSON at line 1, column 17: Expecting value" + ) + + +def test_invalid_plan_encoding_does_not_echo_bytes(tmp_path): + write_plan(tmp_path, "") + (tmp_path / "validation/runtime.json").write_bytes(b"\xffprivate-value") + with pytest.raises(RuntimePlanError) as error: + load_runtime_plan(tmp_path, ExecutionDeclaration()) + assert str(error.value) == "runtime plan contains invalid text encoding" + + @pytest.mark.parametrize("override", ["episode_id", "seed"]) def test_reset_options_cannot_replace_reproducibility_inputs(override): data = plan_data() diff --git a/tests/test_validation/test_runtime_execution.py b/tests/test_validation/test_runtime_execution.py index 6e48c3f7c1..16f13b0ae3 100644 --- a/tests/test_validation/test_runtime_execution.py +++ b/tests/test_validation/test_runtime_execution.py @@ -7,7 +7,7 @@ import pytest from openenv.validation.policy import load_policy, PolicyError -from openenv.validation.providers import StartupError +from openenv.validation.providers import ProviderError, StartupError from openenv.validation.report import CheckResult from openenv.validation.runner import run_validation, source_digest from openenv.validation.runtime.artifacts import write_runtime_bundle @@ -172,6 +172,9 @@ def test_invalid_plan_is_visible_failure(package): provider = FakeRuntimeProvider() report = run_validation(package, max_level=Level.RUNTIME, provider=provider) assert report.verdict.value == "fail" + result = next(r for r in report.results if r.check_id == "runtime.startup") + assert "runtime plan schema validation failed" in result.evidence[0] + assert "missing" in result.evidence[0] assert not provider.builds @@ -185,7 +188,42 @@ def failed_build(*args): report = run_validation(package, max_level=Level.RUNTIME, provider=provider) result = next(r for r in report.results if r.check_id == "runtime.startup") assert result.status is CheckStatus.FAIL + assert result.evidence == ["subject build failed"] + assert report.verdict.value == "fail" + + +def test_provider_failure_diagnostics_are_visible_and_bounded(package): + provider = FakeRuntimeProvider() + + def failed_build(*args): + raise ProviderError("provider deadline elapsed: " + "x" * 5000) + + provider.build = failed_build + report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + result = next(r for r in report.results if r.check_id == "runtime.startup") + assert result.status is CheckStatus.ERROR + assert result.evidence[0].startswith("provider deadline elapsed: ") + assert len(result.evidence[0]) == 4096 + + +def test_source_change_withdraws_dependent_runtime_results(package, monkeypatch): + provider = FakeRuntimeProvider() + + def collect(*args, **kwargs): + (package / "changed.txt").write_text("changed during collection") + return measured_episode() + + monkeypatch.setattr("openenv.validation.runner.collect_runtime_evidence", collect) + report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + results = {result.check_id: result for result in report.results} + assert results["runtime.startup"].status is CheckStatus.ERROR + for name in ("reward_well_formed", "observation_schema", "state_contract"): + result = results[f"runtime.{name}"] + assert result.status is CheckStatus.SKIP + assert "runtime.startup" in result.evidence[0] + assert "source changed" in result.evidence[0] assert report.verdict.value == "fail" + assert provider.subject.stopped @pytest.mark.parametrize("failure", [RuntimeError, KeyboardInterrupt]) diff --git a/tests/test_validation/test_runtime_grading.py b/tests/test_validation/test_runtime_grading.py index d8705f953b..ade9088250 100644 --- a/tests/test_validation/test_runtime_grading.py +++ b/tests/test_validation/test_runtime_grading.py @@ -231,6 +231,28 @@ def test_invalid_raw_envelope_is_rejected_before_defaults(tmp_path, field, value assert "exchange 2: malformed observation envelope" in result.evidence +@pytest.mark.parametrize("data", [None, [], "private-payload", 42, 0.5, False]) +@pytest.mark.parametrize( + "grader,index,envelope", + [ + (ObservationSchemaGrader, 0, "observation"), + (ObservationSchemaGrader, 2, "observation"), + (StateContractGrader, 1, "state"), + (StateContractGrader, 3, "state"), + ], +) +def test_non_object_data_is_a_finding_not_a_validator_crash( + tmp_path, grader, index, envelope, data +): + rows = mutate_response( + good_rows(), index, lambda response: response.update(data=data) + ) + result = grader().run(subject_with(tmp_path, rows)) + assert result.status is CheckStatus.FAIL + assert f"exchange {index}: malformed {envelope} envelope" in result.evidence + assert "private-payload" not in result.model_dump_json() + + def test_missing_done_is_not_filled_by_a_model_default(tmp_path): rows = mutate_response( good_rows(), 2, lambda response: response["data"].pop("done") From 29c58015f49c02ea7b15941a60d5eb3c60a901c3 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Mon, 21 Sep 2026 19:59:58 +0200 Subject: [PATCH 06/29] fix: retain findings when teardown fails --- src/openenv/validation/runner.py | 17 +++-- .../test_validation/test_runtime_execution.py | 66 +++++++++++++++++++ 2 files changed, 77 insertions(+), 6 deletions(-) diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index 6631e44e78..871de1515e 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -234,12 +234,17 @@ def _runtime(subject, *, skip_build, provider): cleanup["completed"] = True except Exception: cleanup["completed"] = False - result = _outcome( - "runtime.startup", - CheckStatus.ERROR, - "subject teardown failed", - started=started, - ) + if result is None: + result = _outcome( + "runtime.startup", + CheckStatus.ERROR, + "subject teardown failed", + started=started, + ) + else: + result.status = CheckStatus.ERROR + result.evidence.append("subject teardown failed") + result.duration_s = time.monotonic() - started return [result, *checks], attempted, plan, evidence, inspection, cleanup diff --git a/tests/test_validation/test_runtime_execution.py b/tests/test_validation/test_runtime_execution.py index 16f13b0ae3..f410deb3be 100644 --- a/tests/test_validation/test_runtime_execution.py +++ b/tests/test_validation/test_runtime_execution.py @@ -11,6 +11,7 @@ from openenv.validation.report import CheckResult from openenv.validation.runner import run_validation, source_digest from openenv.validation.runtime.artifacts import write_runtime_bundle +from openenv.validation.runtime.contracts import RuntimeEvidence from openenv.validation.runtime.scheduler import execute_graders, order_graders from openenv.validation.types import CheckStatus, Level, ProviderCapability from support.runtime import evidence, exchange, FakeRuntimeProvider @@ -240,6 +241,71 @@ def explode(*args, **kwargs): assert "must-not-appear" not in report.model_dump_json() +@pytest.mark.parametrize("failure", [StartupError, ProviderError]) +def test_teardown_failure_preserves_primary_provider_error( + package, monkeypatch, failure +): + provider = FakeRuntimeProvider() + + def failed_inspect(): + raise failure("subject inspection failed") + + def failed_stop(): + raise RuntimeError("token=must-not-appear") + + monkeypatch.setattr(provider.subject, "inspect", failed_inspect) + monkeypatch.setattr(provider.subject, "stop", failed_stop) + report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + result = next(r for r in report.results if r.check_id == "runtime.startup") + assert result.status is CheckStatus.ERROR + assert result.evidence == ["subject inspection failed", "subject teardown failed"] + assert report.verdict.value == "fail" + assert "must-not-appear" not in report.model_dump_json() + + +@pytest.mark.parametrize("collection_failed", [False, True]) +def test_teardown_failure_preserves_collection_outcome( + package, monkeypatch, tmp_path, collection_failed +): + provider = FakeRuntimeProvider() + collected = ( + RuntimeEvidence(failure_phase="schema", failure_reason="schema request failed") + if collection_failed + else measured_episode() + ) + + def failed_stop(): + raise RuntimeError("token=must-not-appear") + + monkeypatch.setattr(provider.subject, "stop", failed_stop) + monkeypatch.setattr( + "openenv.validation.runner.collect_runtime_evidence", lambda *a, **k: collected + ) + bundle = tmp_path / "bundle" + report = run_validation( + package, max_level=Level.RUNTIME, provider=provider, artifacts_dir=bundle + ) + result = next(r for r in report.results if r.check_id == "runtime.startup") + assert result.status is CheckStatus.ERROR + assert result.evidence == [ + "schema request failed" + if collection_failed + else "subject built and reached its control endpoint", + "subject teardown failed", + ] + if not collection_failed: + assert result.measured == { + "provider": provider.name, + "image_ref": "sha256:" + "a" * 64, + } + assert report.verdict.value == "fail" + assert json.loads((bundle / "cleanup.json").read_text()) == { + "required": True, + "completed": False, + } + assert "must-not-appear" not in report.model_dump_json() + + def test_bad_static_bounds_do_not_suppress_independent_runtime(package, monkeypatch): path = package / "openenv.yaml" path.write_text(path.read_text().replace("floor_margin: 0.5", "floor_margin: 0.01")) From ccca191e81212f274ab897275bfd6297fcbcd432 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Tue, 22 Sep 2026 12:39:54 +0200 Subject: [PATCH 07/29] fix: preserve runtime failure reports --- .../validation/graders/runtime/basic.py | 3 + src/openenv/validation/runner.py | 17 +++- .../test_validation/test_runtime_execution.py | 78 +++++++++++++++---- tests/test_validation/test_runtime_grading.py | 2 + 4 files changed, 82 insertions(+), 18 deletions(-) diff --git a/src/openenv/validation/graders/runtime/basic.py b/src/openenv/validation/graders/runtime/basic.py index 34c71d1098..b6700f5d73 100644 --- a/src/openenv/validation/graders/runtime/basic.py +++ b/src/openenv/validation/graders/runtime/basic.py @@ -86,6 +86,9 @@ def check(self, subject, evidence) -> list[str]: continue observations += 1 data = json.loads(exchange.response_json)["data"] + if not isinstance(data, dict): + problems.append(f"exchange {index}: malformed observation envelope") + continue if "reward" not in data: problems.append(f"exchange {index}: missing reward") continue diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index 871de1515e..f9514dea1f 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -232,7 +232,7 @@ def _runtime(subject, *, skip_build, provider): try: running.stop() cleanup["completed"] = True - except Exception: + except (Exception, KeyboardInterrupt): cleanup["completed"] = False if result is None: result = _outcome( @@ -357,12 +357,21 @@ def run_validation( }: reason = "unmet dependency: runtime.startup" results.append(_outcome(entry.check_id, CheckStatus.SKIP, reason)) - if source_digest(target) != digest_before: + try: + digest_after = source_digest(target) + except (ValueError, OSError): + digest_after = None + if digest_after != digest_before: + source_problem = ( + "package source changed during validation" + if digest_after is not None + else "package source could not be verified after validation" + ) results = [ _outcome( r.check_id, CheckStatus.SKIP, - "unmet dependency: runtime.startup (package source changed)", + f"unmet dependency: runtime.startup ({source_problem})", ) if r.check_id in { @@ -378,7 +387,7 @@ def run_validation( _outcome( "runtime.startup", CheckStatus.ERROR, - "package source changed during validation", + source_problem, ) ) diff --git a/tests/test_validation/test_runtime_execution.py b/tests/test_validation/test_runtime_execution.py index f410deb3be..29b0b0b7a5 100644 --- a/tests/test_validation/test_runtime_execution.py +++ b/tests/test_validation/test_runtime_execution.py @@ -2,6 +2,7 @@ import json import shutil +from dataclasses import replace from pathlib import Path from types import SimpleNamespace @@ -207,23 +208,56 @@ def failed_build(*args): assert len(result.evidence[0]) == 4096 -def test_source_change_withdraws_dependent_runtime_results(package, monkeypatch): +@pytest.mark.parametrize("mutation", ["content", "symlink", "unreadable"]) +def test_source_change_withdraws_dependent_runtime_results( + package, monkeypatch, tmp_path, mutation +): provider = FakeRuntimeProvider() + original_digest = source_digest(package) + + def unreadable_source(*args): + raise OSError("private-source-path") def collect(*args, **kwargs): - (package / "changed.txt").write_text("changed during collection") + if mutation == "content": + (package / "changed.txt").write_text("changed during collection") + elif mutation == "symlink": + (package / "changed.txt").symlink_to(tmp_path / "private-source-path") + else: + monkeypatch.setattr( + "openenv.validation.runner.source_digest", unreadable_source + ) return measured_episode() monkeypatch.setattr("openenv.validation.runner.collect_runtime_evidence", collect) - report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + bundle = tmp_path / "bundle" + report = run_validation( + package, max_level=Level.RUNTIME, provider=provider, artifacts_dir=bundle + ) results = {result.check_id: result for result in report.results} assert results["runtime.startup"].status is CheckStatus.ERROR + reason = ( + "package source changed during validation" + if mutation == "content" + else "package source could not be verified after validation" + ) + assert results["runtime.startup"].evidence == [reason] for name in ("reward_well_formed", "observation_schema", "state_contract"): result = results[f"runtime.{name}"] assert result.status is CheckStatus.SKIP assert "runtime.startup" in result.evidence[0] - assert "source changed" in result.evidence[0] + assert reason in result.evidence[0] assert report.verdict.value == "fail" + assert report.source_digest == original_digest + assert json.loads((bundle / "report.json").read_text()) == report.model_dump( + mode="json" + ) + assert json.loads((bundle / "cleanup.json").read_text()) == { + "required": True, + "completed": True, + } + assert len(json.loads((bundle / "collector-trace.json").read_text())) == 4 + assert "private-source-path" not in report.model_dump_json() assert provider.subject.stopped @@ -263,19 +297,27 @@ def failed_stop(): assert "must-not-appear" not in report.model_dump_json() -@pytest.mark.parametrize("collection_failed", [False, True]) +@pytest.mark.parametrize("collection_state", ["complete", "failed", "partial"]) +@pytest.mark.parametrize("teardown_error", [RuntimeError, KeyboardInterrupt]) def test_teardown_failure_preserves_collection_outcome( - package, monkeypatch, tmp_path, collection_failed + package, monkeypatch, tmp_path, collection_state, teardown_error ): provider = FakeRuntimeProvider() - collected = ( - RuntimeEvidence(failure_phase="schema", failure_reason="schema request failed") - if collection_failed - else measured_episode() - ) + collected = measured_episode() + if collection_state == "failed": + collected = RuntimeEvidence( + failure_phase="schema", failure_reason="schema request failed" + ) + elif collection_state == "partial": + collected = replace( + collected, + exchanges=collected.exchanges[:2], + failure_phase="step", + failure_reason="step request failed", + ) def failed_stop(): - raise RuntimeError("token=must-not-appear") + raise teardown_error("token=must-not-appear") monkeypatch.setattr(provider.subject, "stop", failed_stop) monkeypatch.setattr( @@ -289,11 +331,11 @@ def failed_stop(): assert result.status is CheckStatus.ERROR assert result.evidence == [ "schema request failed" - if collection_failed + if collection_state == "failed" else "subject built and reached its control endpoint", "subject teardown failed", ] - if not collection_failed: + if collection_state != "failed": assert result.measured == { "provider": provider.name, "image_ref": "sha256:" + "a" * 64, @@ -303,6 +345,14 @@ def failed_stop(): "required": True, "completed": False, } + assert len(json.loads((bundle / "collector-trace.json").read_text())) == len( + collected.exchanges + ) + saved_evidence = json.loads((bundle / "collector-evidence.json").read_text()) + assert saved_evidence["failure_reason"] == collected.failure_reason + assert json.loads((bundle / "report.json").read_text()) == report.model_dump( + mode="json" + ) assert "must-not-appear" not in report.model_dump_json() diff --git a/tests/test_validation/test_runtime_grading.py b/tests/test_validation/test_runtime_grading.py index ade9088250..9df9609778 100644 --- a/tests/test_validation/test_runtime_grading.py +++ b/tests/test_validation/test_runtime_grading.py @@ -235,6 +235,8 @@ def test_invalid_raw_envelope_is_rejected_before_defaults(tmp_path, field, value @pytest.mark.parametrize( "grader,index,envelope", [ + (RewardWellFormedGrader, 0, "observation"), + (RewardWellFormedGrader, 2, "observation"), (ObservationSchemaGrader, 0, "observation"), (ObservationSchemaGrader, 2, "observation"), (StateContractGrader, 1, "state"), From 7e9b888c6b0559bbf0078763f97a6851d1c4d800 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Tue, 22 Sep 2026 14:02:14 +0200 Subject: [PATCH 08/29] refactor: simplify runtime validation --- src/openenv/validation/runner.py | 16 ++-------- src/openenv/validation/runtime/scheduler.py | 30 ++++++------------- .../test_validation/test_runtime_transport.py | 6 ---- 3 files changed, 11 insertions(+), 41 deletions(-) diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index f9514dea1f..cc4071bfce 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -108,6 +108,7 @@ def _runtime(subject, *, skip_build, provider): started = time.monotonic() attempted = False result = None + checks = [] try: if skip_build: raise UnsupportedCapability( @@ -184,23 +185,13 @@ def _runtime(subject, *, skip_build, provider): result = _outcome( "runtime.startup", CheckStatus.SKIP, str(exc), started=started ) - checks = [] - except RuntimePlanError as exc: + except (RuntimePlanError, StartupError) as exc: result = _outcome( "runtime.startup", CheckStatus.FAIL, str(exc)[:4096], started=started, ) - checks = [] - except StartupError as exc: - result = _outcome( - "runtime.startup", - CheckStatus.FAIL, - str(exc)[:4096], - started=started, - ) - checks = [] except ProviderError as exc: result = _outcome( "runtime.startup", @@ -208,7 +199,6 @@ def _runtime(subject, *, skip_build, provider): str(exc)[:4096], started=started, ) - checks = [] except KeyboardInterrupt as exc: if isinstance(exc, RuntimeCollectionInterrupted): evidence = exc.evidence @@ -218,7 +208,6 @@ def _runtime(subject, *, skip_build, provider): "validation interrupted", started=started, ) - checks = [] except Exception as exc: result = _outcome( "runtime.startup", @@ -226,7 +215,6 @@ def _runtime(subject, *, skip_build, provider): f"runtime orchestration failed ({type(exc).__name__})", started=started, ) - checks = [] finally: if running is not None: try: diff --git a/src/openenv/validation/runtime/scheduler.py b/src/openenv/validation/runtime/scheduler.py index 012cabba17..7f9cfb39ac 100644 --- a/src/openenv/validation/runtime/scheduler.py +++ b/src/openenv/validation/runtime/scheduler.py @@ -1,6 +1,7 @@ """Deterministic dependency ordering and capability-aware grader execution.""" import time +from graphlib import CycleError, TopologicalSorter from ..policy import PolicyError from ..report import CheckResult @@ -12,27 +13,14 @@ def order_graders(graders: list) -> list: by_id = {grader.check_id: grader for grader in graders} if len(by_id) != len(graders): raise PolicyError("duplicate grader IDs") - visiting = set() - visited = set() - ordered = [] - - def visit(check_id): - if check_id in visiting: - raise PolicyError(f"grader dependency cycle at {check_id}") - if check_id in visited: - return - visiting.add(check_id) - grader = by_id[check_id] - for dependency in sorted(grader.depends_on): - if dependency in by_id: - visit(dependency) - visiting.remove(check_id) - visited.add(check_id) - ordered.append(grader) - - for check_id in sorted(by_id): - visit(check_id) - return ordered + graph = { + check_id: sorted(dep for dep in grader.depends_on if dep in by_id) + for check_id, grader in sorted(by_id.items()) + } + try: + return [by_id[check_id] for check_id in TopologicalSorter(graph).static_order()] + except CycleError as exc: + raise PolicyError(f"grader dependency cycle at {exc.args[1][0]}") from exc def execute_graders(graders, subject, *, provider_capabilities=frozenset(), prior=()): diff --git a/tests/test_validation/test_runtime_transport.py b/tests/test_validation/test_runtime_transport.py index b26b3d2393..06929bcc8c 100644 --- a/tests/test_validation/test_runtime_transport.py +++ b/tests/test_validation/test_runtime_transport.py @@ -33,12 +33,6 @@ def __init__(self, *, close_send=False, close_failure=False, block_send=False): self.operation = None self.steps = 0 - def __enter__(self): - return self - - def __exit__(self, *_args): - self.close() - def send(self, payload): self.operation = json.loads(payload)["type"] if self.block_send: From 65be5dd89666a01d30bbeabf4d1d1d9cf1e98bf5 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 23 Sep 2026 06:28:59 +0000 Subject: [PATCH 09/29] fix: report initial source digest failures --- src/openenv/validation/runner.py | 24 ++++++++++++++++++++---- tests/test_validation/test_runner.py | 18 ++++++++++++++++++ 2 files changed, 38 insertions(+), 4 deletions(-) diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index cc4071bfce..a1fae776af 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -280,7 +280,10 @@ def run_validation( "runtime validation requires policy v2; v1 supports --level static" ) signature = detect_signature(target) - digest_before = source_digest(target) + try: + digest_before = source_digest(target) + except (ValueError, OSError): + digest_before = "" parser = default_parser_registry().parser_for(signature) manifest: NormalizedManifest | None = None @@ -324,6 +327,15 @@ def run_validation( if attempted: levels.append(Level.RUNTIME) + if not digest_before: + static_result = next( + result for result in results if result.check_id == "static.manifest" + ) + static_result.status = CheckStatus.ERROR + static_result.evidence.append( + "package source could not be verified before validation" + ) + if wants_runtime: # Policy IDs are an inventory, not evidence that a grader exists. Keep the # incomplete surface explicit throughout the staged implementation. @@ -351,9 +363,13 @@ def run_validation( digest_after = None if digest_after != digest_before: source_problem = ( - "package source changed during validation" - if digest_after is not None - else "package source could not be verified after validation" + "package source could not be verified before validation" + if not digest_before + else ( + "package source changed during validation" + if digest_after is not None + else "package source could not be verified after validation" + ) ) results = [ _outcome( diff --git a/tests/test_validation/test_runner.py b/tests/test_validation/test_runner.py index 8117819033..8f5fa70793 100644 --- a/tests/test_validation/test_runner.py +++ b/tests/test_validation/test_runner.py @@ -87,3 +87,21 @@ def test_source_digest_uses_portable_relative_paths(tmp_path): expected = hashlib.sha256(b"nested/file.txt\0contents\0").hexdigest() assert source_digest(package_root) == expected + + +@pytest.mark.parametrize("failure", [ValueError, OSError]) +def test_initial_source_digest_failure_is_reported(monkeypatch, failure): + def fail_digest(*args): + raise failure("private-source-path") + + monkeypatch.setattr("openenv.validation.runner.source_digest", fail_digest) + report = run_validation(FIXTURES / "served_min_pass", max_level=Level.STATIC) + + (result,) = report.results + assert report.source_digest == "" + assert result.status is CheckStatus.ERROR + assert ( + result.evidence[-1] == "package source could not be verified before validation" + ) + assert report.verdict is Verdict.FAIL + assert "private-source-path" not in report.model_dump_json() From 4c746aa570732346d06f096beb51c20ddca0d185 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 23 Sep 2026 06:35:34 +0000 Subject: [PATCH 10/29] fix: stop validation after source digest failure --- src/openenv/validation/runner.py | 60 ++++++++++++++-------------- tests/test_validation/test_runner.py | 6 +++ 2 files changed, 37 insertions(+), 29 deletions(-) diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index a1fae776af..a3c96e0fed 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -290,19 +290,29 @@ def run_validation( results: list[CheckResult] = [] parse_started = time.monotonic() - try: - manifest = parser.parse(target) - except ManifestError as exc: + if not digest_before: results.append( - CheckResult( - check_id="static.manifest", - status=CheckStatus.FAIL, - measured={"schema_errors": len(exc.errors)}, - evidence=exc.errors, - remediation=exc.remediation, - duration_s=time.monotonic() - parse_started, + _outcome( + "static.manifest", + CheckStatus.ERROR, + "package source could not be verified before validation", + started=parse_started, ) ) + else: + try: + manifest = parser.parse(target) + except ManifestError as exc: + results.append( + CheckResult( + check_id="static.manifest", + status=CheckStatus.FAIL, + measured={"schema_errors": len(exc.errors)}, + evidence=exc.errors, + remediation=exc.remediation, + duration_s=time.monotonic() - parse_started, + ) + ) levels = [Level.STATIC] plan = evidence = None @@ -327,15 +337,6 @@ def run_validation( if attempted: levels.append(Level.RUNTIME) - if not digest_before: - static_result = next( - result for result in results if result.check_id == "static.manifest" - ) - static_result.status = CheckStatus.ERROR - static_result.evidence.append( - "package source could not be verified before validation" - ) - if wants_runtime: # Policy IDs are an inventory, not evidence that a grader exists. Keep the # incomplete surface explicit throughout the staged implementation. @@ -357,20 +358,21 @@ def run_validation( }: reason = "unmet dependency: runtime.startup" results.append(_outcome(entry.check_id, CheckStatus.SKIP, reason)) - try: - digest_after = source_digest(target) - except (ValueError, OSError): - digest_after = None - if digest_after != digest_before: - source_problem = ( - "package source could not be verified before validation" - if not digest_before - else ( + source_problem = None + if not digest_before: + source_problem = "package source could not be verified before validation" + else: + try: + digest_after = source_digest(target) + except (ValueError, OSError): + digest_after = None + if digest_after != digest_before: + source_problem = ( "package source changed during validation" if digest_after is not None else "package source could not be verified after validation" ) - ) + if source_problem is not None: results = [ _outcome( r.check_id, diff --git a/tests/test_validation/test_runner.py b/tests/test_validation/test_runner.py index 8f5fa70793..961261d6e8 100644 --- a/tests/test_validation/test_runner.py +++ b/tests/test_validation/test_runner.py @@ -94,7 +94,13 @@ def test_initial_source_digest_failure_is_reported(monkeypatch, failure): def fail_digest(*args): raise failure("private-source-path") + def fail_parse(*args): + raise AssertionError("rejected source must not be parsed") + monkeypatch.setattr("openenv.validation.runner.source_digest", fail_digest) + monkeypatch.setattr( + "openenv.validation.parsers.openenv_yaml.OpenEnvYamlParser.parse", fail_parse + ) report = run_validation(FIXTURES / "served_min_pass", max_level=Level.STATIC) (result,) = report.results From d94f0218e5b3f5cd5c348b0cf3df25ece67ee245 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 23 Sep 2026 08:18:59 +0000 Subject: [PATCH 11/29] fix: preserve runtime evidence on close interrupt --- src/openenv/validation/runtime/collector.py | 4 +- .../test_validation/test_runtime_transport.py | 48 +++++++++++++++++++ 2 files changed, 50 insertions(+), 2 deletions(-) diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index 1160dbbcd8..f9d9b32aaa 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -216,11 +216,11 @@ def exchange(operation: str, data: dict | None = None) -> dict: lambda: connection.send(json.dumps({"type": "close"})), min(remaining(), 1.0), ) - except Exception: + except (Exception, KeyboardInterrupt): pass # A server may close immediately after its final response. try: _bounded_call(connection, connection.close, 1.0) - except Exception: + except (Exception, KeyboardInterrupt): _abort_transport(connection) return RuntimeEvidence( exchanges=tuple(exchanges), observation_schema_json=schema_json diff --git a/tests/test_validation/test_runtime_transport.py b/tests/test_validation/test_runtime_transport.py index 06929bcc8c..dc42111a58 100644 --- a/tests/test_validation/test_runtime_transport.py +++ b/tests/test_validation/test_runtime_transport.py @@ -121,6 +121,38 @@ def test_teardown_noise_does_not_fail_a_completed_episode(collect, fault): assert connection.closed +@pytest.mark.parametrize("fault", ["close_send", "close"]) +def test_teardown_interrupt_does_not_fail_a_completed_episode(collect, fault): + connection = EpisodeConnection() + if fault == "close_send": + send = connection.send + + def interrupt_close_send(payload): + send(payload) + if connection.operation == "close": + raise KeyboardInterrupt + + connection.send = interrupt_close_send + else: + + def interrupt_close(): + connection.closed = True + raise KeyboardInterrupt + + connection.close = interrupt_close + + evidence = collect(connection) + assert [row.operation for row in evidence.exchanges] == [ + "reset", + "state", + "step", + "state", + ] + assert evidence.failure_reason is None + assert evidence.failure_phase is None + assert connection.closed + + def test_teardown_failure_does_not_replace_a_genuine_operation_failure(collect): connection = EpisodeConnection(close_failure=True) connection.step_failure = True @@ -131,6 +163,22 @@ def test_teardown_failure_does_not_replace_a_genuine_operation_failure(collect): assert connection.closed +def test_teardown_interrupt_does_not_replace_a_genuine_operation_failure(collect): + connection = EpisodeConnection() + connection.step_failure = True + + def interrupt_close(): + connection.closed = True + raise KeyboardInterrupt + + connection.close = interrupt_close + evidence = collect(connection) + assert evidence.failure_phase == "step" + assert evidence.failure_reason == "step failed (ValueError)" + assert [row.operation for row in evidence.exchanges] == ["reset", "state"] + assert connection.closed + + def test_interruption_preserves_partial_evidence_despite_teardown_failure(collect): connection = EpisodeConnection(close_failure=True) receive = connection.recv From 434bb1dcf892119b1d7344c7fc1c1459a7da390a Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Thu, 24 Sep 2026 12:09:30 +0200 Subject: [PATCH 12/29] feat: add session validation telemetry --- rfcs/008-environment-auto-validation.md | 30 +- src/openenv/core/env_server/http_server.py | 108 +++++++ .../core/env_server/session_telemetry.py | 241 +++++++++++++++ src/openenv/core/rubrics/base.py | 12 + src/openenv/validation/runner.py | 3 + src/openenv/validation/runtime/artifacts.py | 12 + src/openenv/validation/runtime/collector.py | 102 ++++++- src/openenv/validation/runtime/contracts.py | 6 + .../validation/runtime/served_probe/app.py | 14 +- .../integration/test_runtime_process.py | 219 ++++++++++++++ .../test_session_telemetry_protocol.py | 275 ++++++++++++++++++ .../test_validation/test_runtime_artifacts.py | 14 + .../test_validation/test_runtime_collector.py | 12 + .../test_runtime_telemetry_collector.py | 111 +++++++ .../test_validation/test_session_telemetry.py | 102 +++++++ tests/validation_runtime/README.md | 7 + tests/validation_runtime/acceptance.json | 20 +- 17 files changed, 1282 insertions(+), 6 deletions(-) create mode 100644 src/openenv/core/env_server/session_telemetry.py create mode 100644 tests/test_validation/integration/test_runtime_process.py create mode 100644 tests/test_validation/integration/test_session_telemetry_protocol.py create mode 100644 tests/test_validation/test_runtime_telemetry_collector.py create mode 100644 tests/test_validation/test_session_telemetry.py diff --git a/rfcs/008-environment-auto-validation.md b/rfcs/008-environment-auto-validation.md index 63405af7e6..6ad227f5f0 100644 --- a/rfcs/008-environment-auto-validation.md +++ b/rfcs/008-environment-auto-validation.md @@ -590,7 +590,35 @@ attached to the **same** replay connection, reject unauthorized/cross-session reads and never expose telemetry as agent MCP tools. A second WebSocket creates another environment and cannot supply evidence for the measured instance. A validator transcript alone cannot pass subject-emitted trajectory recording. -These authorization requirements do not add new public wire messages in this slice. +The initial contracts slice added no public wire messages. PR4 implements the +following opt-in protocol. + +#### Session telemetry protocol (PR4) + +The server opts in only when `OPENENV_VALIDATION_TOKEN` is provisioned explicitly +by the validation supervisor. On its existing simulation `/ws` connection, the +collector sends `validation_open` with `data: {schema_version: 1, token: ...}`. +The reply returns a random capability bound to that connection. Subsequent +`validation_read` messages carry that capability and return a `validation` +snapshot. Disabled, unauthorized and cross-connection requests fail without +echoing credentials. Closing the connection destroys the capability. MCP and +production endpoints never expose these operations. Authentication exchanges +are excluded from persisted evidence. + +Snapshots identify requested and actually forwarded seed arguments; successful +reset alone is not acceptance. They include a named rubric tree rooted at `root`, +explicit safe configuration, and per-step attribution with operation-local +evaluation flags. Unevaluated gated children never reuse an earlier score. +Stock container aggregation is named explicitly; custom rubrics may supply +`validation_config()` to expose public JSON configuration. Arbitrary attributes +and `state_dict()` are never serialized as configuration. + +The subject server also emits a bounded record of the reset/step/state request +and response envelopes it executed. The validator captures the wire separately +and later compares the two. The record is bound to the authenticated session, +limited to 100 steps/202 operations and 8 MiB, and marks truncation explicitly. +Missing configuration, attribution or records cannot be inferred from other +successful operations. This transport introduces no new passing grader by itself. Applicability predicates must distinguish empty declared sets from absent capabilities. Missing subject features, missing provider support and checks whose diff --git a/src/openenv/core/env_server/http_server.py b/src/openenv/core/env_server/http_server.py index 50dd5b1746..2e80677fd7 100644 --- a/src/openenv/core/env_server/http_server.py +++ b/src/openenv/core/env_server/http_server.py @@ -14,6 +14,7 @@ import json import logging import os +import secrets import time import uuid from concurrent.futures import ThreadPoolExecutor @@ -61,6 +62,17 @@ ) from .route_config import GetEndpointConfig, register_get_endpoints from .serialization import deserialize_action, serialize_observation +from .session_telemetry import ( + rubric_counts, + rubric_snapshot, + SeedAcceptance, + SessionTelemetry, + ValidationOpenedData, + ValidationOpenedResponse, + ValidationOpenMessage, + ValidationReadMessage, + ValidationResponse, +) from .types import ( Action, ConcurrencyConfig, @@ -667,6 +679,12 @@ def register_routes( f"Invalid mode: '{mode}'. Must be one of: {valid_modes}" ) + # Only explicitly provisioned simulation servers accept validation controls. + validation_token = os.environ.get("OPENENV_VALIDATION_TOKEN", "") + validation_enabled = ( + mode == ServerMode.SIMULATION and 32 <= len(validation_token) <= 256 + ) + # Wire up idle-session reaper lifecycle via app events server_ref = self @@ -1534,6 +1552,8 @@ async def websocket_endpoint(websocket: WebSocket): session_env = None owns_session = False attached_session = False + telemetry = None + operations_started = False try: requested_session_id = websocket.query_params.get("session_id") @@ -1598,6 +1618,64 @@ async def websocket_endpoint(websocket: WebSocket): msg_type = message_dict.get("type", "") + if msg_type in {"validation_open", "validation_read"}: + # Do not return Pydantic input/error details: these messages + # contain credentials and must never echo or log them. + try: + if not validation_enabled or not owns_session: + raise ValueError("Validation unavailable") + if msg_type == "validation_open": + auth = ValidationOpenMessage.model_validate( + message_dict + ) + if telemetry is not None or operations_started: + raise ValueError("Validation already started") + if not secrets.compare_digest( + validation_token.encode(), + auth.data.token.get_secret_value().encode(), + ): + raise ValueError("Unauthorized") + telemetry = SessionTelemetry() + response = ValidationOpenedResponse( + data=ValidationOpenedData( + capability=telemetry.capability + ) + ) + else: + auth = ValidationReadMessage.model_validate( + message_dict + ) + if telemetry is None or not telemetry.authorized( + auth.data.capability.get_secret_value() + ): + raise ValueError("Unauthorized") + response = ValidationResponse( + data=telemetry.snapshot + ) + except Exception: + response = WSErrorResponse( + data={ + "message": "Validation unavailable or unauthorized", + "code": WSErrorCode.VALIDATION_ERROR, + } + ) + await websocket.send_text(response.model_dump_json()) + continue + + seed_acceptance = None + before_scores = None + if msg_type in {"reset", "step", "state"}: + operations_started = True + if telemetry is not None and msg_type == "step": + try: + before_scores = rubric_counts( + getattr(session_env, "rubric", None) + ) + except Exception: + telemetry.snapshot.rubric_error = ( + "Rubric introspection unavailable" + ) + try: match msg_type: case "reset": @@ -1629,6 +1707,12 @@ async def websocket_endpoint(websocket: WebSocket): ) ) + if telemetry is not None: + seed_acceptance = SeedAcceptance( + requested="seed" in msg.data, + value=msg.data.get("seed"), + accepted="seed" in valid_kwargs, + ) self._update_session_activity(session_id) response = WSObservationResponse( @@ -1706,6 +1790,30 @@ async def websocket_endpoint(websocket: WebSocket): } ) + if telemetry is not None and msg_type in { + "reset", + "step", + "state", + }: + nodes = None + try: + if msg_type in {"reset", "step"}: + nodes = rubric_snapshot( + getattr(session_env, "rubric", None), + before_scores, + ) + except Exception: + nodes = [] + telemetry.snapshot.rubric_error = ( + "Rubric introspection unavailable" + ) + telemetry.append( + msg_type, + message_dict, + response.model_dump(mode="json"), + seed=seed_acceptance, + rubric=nodes, + ) await websocket.send_text(response.model_dump_json()) except ValidationError as e: diff --git a/src/openenv/core/env_server/session_telemetry.py b/src/openenv/core/env_server/session_telemetry.py new file mode 100644 index 0000000000..fe9c3eb0f4 --- /dev/null +++ b/src/openenv/core/env_server/session_telemetry.py @@ -0,0 +1,241 @@ +# SPDX-License-Identifier: BSD-3-Clause + +"""Bounded, opt-in evidence emitted by the subject's replay session.""" + +import json +import secrets +from typing import Any, Literal + +from openenv.core.rubrics.base import Rubric +from openenv.core.rubrics.containers import Gate, Sequential, WeightedSum +from pydantic import BaseModel, ConfigDict, Field, SecretStr + +MAX_TELEMETRY_BYTES = 8 * 1024 * 1024 +MAX_TRAJECTORY_ACTIONS = 100 +MAX_TRAJECTORY_RECORDS = 202 +MAX_RUBRIC_NODES = 128 + + +class _WireModel(BaseModel): + model_config = ConfigDict(extra="forbid", strict=True) + + +class ValidationOpenData(_WireModel): + schema_version: Literal[1] + token: SecretStr = Field(min_length=32, max_length=256) + + +class ValidationOpenMessage(_WireModel): + type: Literal["validation_open"] + data: ValidationOpenData + + +class ValidationReadData(_WireModel): + schema_version: Literal[1] + capability: SecretStr = Field(min_length=32, max_length=256) + + +class ValidationReadMessage(_WireModel): + type: Literal["validation_read"] + data: ValidationReadData + + +class ValidationOpenedData(_WireModel): + schema_version: Literal[1] = 1 + capability: str + + +class ValidationOpenedResponse(_WireModel): + type: Literal["validation_open"] = "validation_open" + data: ValidationOpenedData + + +class SeedAcceptance(_WireModel): + requested: bool + value: Any = None + accepted: bool + + +class RubricNode(_WireModel): + name: str + class_name: str + children: list[str] + aggregation: Literal["weighted_sum", "sequential", "gate", "leaf", "unknown"] + config: dict[str, Any] + config_available: bool + score: Any = None + evaluated: bool = False + + +class StepAttribution(_WireModel): + step_index: int + rubric: list[RubricNode] + + +class TrajectoryRecord(_WireModel): + operation: Literal["reset", "step", "state"] + request: dict[str, Any] + response: dict[str, Any] + + +class SubjectTrajectory(_WireModel): + schema_version: Literal[1] = 1 + source: Literal["openenv-server"] = "openenv-server" + records: list[TrajectoryRecord] = Field(default_factory=list) + complete: bool = True + reason: str | None = None + + +class ValidationSnapshot(_WireModel): + schema_version: Literal[1] = 1 + seed: SeedAcceptance | None = None + rubric: list[RubricNode] = Field(default_factory=list) + rubric_error: str | None = None + attribution: list[StepAttribution] = Field(default_factory=list) + trajectory: SubjectTrajectory = Field(default_factory=SubjectTrajectory) + + +class ValidationResponse(_WireModel): + type: Literal["validation"] = "validation" + data: ValidationSnapshot + + +def _rubrics(root: Rubric | None) -> list[tuple[str, Rubric]]: + """Visit the named tree with finite size, rejecting cycles and aliases.""" + if root is None: + return [] + result, pending, seen = [], [("root", root)], set() + while pending: + name, rubric = pending.pop() + if id(rubric) in seen or len(result) >= MAX_RUBRIC_NODES: + raise ValueError("Rubric tree is cyclic, shared, or exceeds 128 nodes") + seen.add(id(rubric)) + result.append((name, rubric)) + children = list(rubric.named_children()) + if any(not key or "." in key for key, _ in children): + raise ValueError("Rubric child names must be nonempty path segments") + pending.extend((f"{name}.{key}", child) for key, child in reversed(children)) + return result + + +def rubric_counts(root: Rubric | None) -> dict[int, int]: + return {id(r): r._evaluation_count for _, r in _rubrics(root)} + + +def rubric_snapshot( + root: Rubric | None, before: dict[int, int] | None = None +) -> list[RubricNode]: + nodes = [] + for name, rubric in _rubrics(root): + children = [f"{name}.{key}" for key, _ in rubric.named_children()] + aggregation = "unknown" if children else "leaf" + config = None + # Exact types: a subclass can override scoring, so its semantics are unknown. + if type(rubric) is WeightedSum: + aggregation, config = "weighted_sum", {"weights": list(rubric._weights)} + elif type(rubric) is Sequential: + aggregation, config = "sequential", {} + elif type(rubric) is Gate: + aggregation, config = "gate", {"threshold": rubric.threshold} + else: + config = rubric.validation_config() + evaluated = before is not None and rubric._evaluation_count > before.get( + id(rubric), 0 + ) + nodes.append( + RubricNode( + name=name, + class_name=f"{type(rubric).__module__}.{type(rubric).__qualname__}", + children=children, + aggregation=aggregation, + config={} if config is None else config, + config_available=config is not None, + score=rubric.last_score if evaluated else None, + evaluated=evaluated, + ) + ) + json.dumps([node.model_dump() for node in nodes], allow_nan=False) + return nodes + + +class SessionTelemetry: + """Own one socket's capability and a detached record of completed operations.""" + + def __init__(self): + self.capability = secrets.token_urlsafe(32) + self.snapshot = ValidationSnapshot() + self._bytes = 0 + self._steps = 0 + + def authorized(self, capability: str) -> bool: + return secrets.compare_digest(self.capability.encode(), capability.encode()) + + def append( + self, + operation: Literal["reset", "step", "state"], + request: dict[str, Any], + response: dict[str, Any], + *, + seed: SeedAcceptance | None = None, + rubric: list[RubricNode] | None = None, + ) -> None: + if not self.snapshot.trajectory.complete: + return + try: + # JSON round-trip detaches mutable observations/actions from the record. + record = TrajectoryRecord.model_validate( + json.loads( + json.dumps( + { + "operation": operation, + "request": request, + "response": response, + }, + allow_nan=False, + ) + ) + ) + attribution = ( + StepAttribution(step_index=self._steps, rubric=rubric or []) + if operation == "step" + else None + ) + byte_count = len(record.model_dump_json().encode()) + if attribution is not None: + byte_count += len(attribution.model_dump_json().encode()) + byte_count += ( + len(seed.model_dump_json().encode()) if seed is not None else 0 + ) + # Reserve room for the current rubric, envelope, and seed metadata. + current_bytes = len( + json.dumps( + [ + node.model_dump() + for node in (self.snapshot.rubric if rubric is None else rubric) + ], + allow_nan=False, + ).encode() + ) + if ( + len(self.snapshot.trajectory.records) >= MAX_TRAJECTORY_RECORDS + or operation == "step" + and self._steps >= MAX_TRAJECTORY_ACTIONS + or self._bytes + byte_count + current_bytes + 4096 > MAX_TELEMETRY_BYTES + ): + self.incomplete("Session telemetry budget exceeded") + return + self._bytes += byte_count + self.snapshot.trajectory.records.append(record) + if seed is not None: + self.snapshot.seed = seed.model_copy(deep=True) + if rubric is not None: + self.snapshot.rubric = [node.model_copy(deep=True) for node in rubric] + if attribution is not None: + self.snapshot.attribution.append(attribution.model_copy(deep=True)) + self._steps += 1 + except (TypeError, ValueError): + self.incomplete("Session telemetry contains non-JSON data") + + def incomplete(self, reason: str) -> None: + self.snapshot.trajectory.complete = False + self.snapshot.trajectory.reason = reason diff --git a/src/openenv/core/rubrics/base.py b/src/openenv/core/rubrics/base.py index 1ba184bfda..fc0786ccc1 100644 --- a/src/openenv/core/rubrics/base.py +++ b/src/openenv/core/rubrics/base.py @@ -46,6 +46,7 @@ def __init__(self): object.__setattr__(self, "_forward_hooks", []) object.__setattr__(self, "_forward_pre_hooks", []) object.__setattr__(self, "last_score", None) + object.__setattr__(self, "_evaluation_count", 0) def __setattr__(self, name: str, value: Any) -> None: # Auto-register child rubrics when assigned as attributes @@ -95,6 +96,7 @@ async def _run_forward_pre_hooks_async(self, action: Any, observation: Any) -> N def _finish_forward(self, action: Any, observation: Any, result: float) -> float: """Store the result and run post-forward hooks synchronously.""" self.last_score = result + self._evaluation_count += 1 # Post-forward hooks for hook in self._forward_hooks: @@ -107,6 +109,7 @@ async def _finish_forward_async( ) -> float: """Store the result and run post-forward hooks from an async call path.""" self.last_score = result + self._evaluation_count += 1 # Post-forward hooks for hook in self._forward_hooks: @@ -207,6 +210,15 @@ def reset(self) -> None: """Reset any internal state. Override in subclasses if needed.""" pass + def validation_config(self) -> Optional[Dict[str, Any]]: + """Return explicitly public configuration for authorized validation. + + Override to expose JSON configuration without credentials or runtime state. + ``None`` means configuration introspection is unavailable. Stock aggregation + containers expose their weights/threshold directly through the server. + """ + return None + def state_dict(self) -> Dict[str, Any]: """Serialize rubric configuration for checkpointing.""" return {} diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index a3c96e0fed..8151dbc133 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -2,6 +2,7 @@ import hashlib import os +import secrets import stat import time import uuid @@ -140,6 +141,7 @@ def _runtime(subject, *, skip_build, provider): resources=manifest.resources, network=manifest.network, run_id="validation-" + uuid.uuid4().hex, + env_vars={"OPENENV_VALIDATION_TOKEN": secrets.token_urlsafe(32)}, ) attempted = True image_ref = provider.build(subject.root, manifest.execution) @@ -159,6 +161,7 @@ def _runtime(subject, *, skip_build, provider): running.base_url, plan, episode_timeout_s=manifest.resources.episode_timeout_s, + validation_token=spec.env_vars["OPENENV_VALIDATION_TOKEN"], ) # A health endpoint without a functioning protocol isn't a startup success. if not evidence.exchanges and evidence.failure_reason: diff --git a/src/openenv/validation/runtime/artifacts.py b/src/openenv/validation/runtime/artifacts.py index d2d8e340bb..44585100d4 100644 --- a/src/openenv/validation/runtime/artifacts.py +++ b/src/openenv/validation/runtime/artifacts.py @@ -101,10 +101,22 @@ def write_runtime_bundle( "failure_reason": evidence.failure_reason, "complete": evidence.failure_phase is None and evidence.failure_reason is None, + "telemetry_error": evidence.telemetry_error, } + telemetry = None + telemetry_parse_failed = False + if evidence.telemetry_json is not None: + try: + telemetry = json.loads(evidence.telemetry_json) + files["session-telemetry.json"] = telemetry + except (ValueError, RecursionError): + telemetry_parse_failed = True + collector_metadata["telemetry_error"] = "malformed telemetry omitted" collector_metadata["redacted"] = ( bool(omitted_fields) or schema_parse_failed + or telemetry_parse_failed + or _redact(telemetry) != telemetry or _redact(trace) != trace or _redact(collector_metadata) != collector_metadata ) diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index f9d9b32aaa..5f1095e7a2 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -72,6 +72,7 @@ def collect_runtime_evidence( *, episode_timeout_s: float, request_timeout_s: float = 5.0, + validation_token: str | None = None, ) -> RuntimeEvidence: """ Preserve schema and reset/step/state responses without model coercion. @@ -89,6 +90,8 @@ def collect_runtime_evidence( Deadline for the complete collection, including schema retrieval. request_timeout_s (`float`, *optional*, defaults to `5.0`): Per-operation deadline, capped by the remaining episode budget. + validation_token (`str`, *optional*): + Run-scoped telemetry authorization; never retained in evidence. Returns: [`~openenv.validation.runtime.contracts.RuntimeEvidence`]: raw evidence. @@ -98,6 +101,8 @@ def collect_runtime_evidence( schema_json = None phase = "schema" trace_bytes = 0 + telemetry_json = None + telemetry_error = None def remaining() -> float: value = min(request_timeout_s, deadline - time.monotonic()) @@ -131,12 +136,15 @@ def remaining() -> float: schema = json.loads(payload) if not isinstance(schema, dict) or "observation" not in schema: raise ValueError("missing observation schema") - schema_json = json.dumps( + schema_payload = json.dumps( schema["observation"], allow_nan=False, ensure_ascii=False, separators=(",", ":"), ) + if validation_token and validation_token in schema_payload: + raise ValueError("schema contains validation credentials") + schema_json = schema_payload endpoint = urlsplit(base_url) ws_url = urlunsplit( @@ -154,13 +162,59 @@ def remaining() -> float: proxy=None, open_timeout=remaining(), close_timeout=1, - max_size=MAX_MESSAGE_BYTES, + max_size=MAX_TRACE_BYTES if validation_token else MAX_MESSAGE_BYTES, max_queue=1, compression=None, ) complete = False try: + def telemetry_request(operation, data): + request = json.dumps({"type": operation, "data": data}) + _bounded_call(connection, lambda: connection.send(request), remaining()) + raw = connection.recv(timeout=remaining()) + if not isinstance(raw, str) or len(raw.encode()) > MAX_TRACE_BYTES: + raise ValueError("invalid telemetry response") + response = json.loads(raw) + if not isinstance(response, dict): + raise ValueError("invalid telemetry envelope") + return response + + capability = None + if validation_token: + phase = "validation_open" + response = telemetry_request( + phase, {"schema_version": 1, "token": validation_token} + ) + data = response.get("data") + if ( + response.get("type") == "validation_open" + and isinstance(data, dict) + and data.get("schema_version") == 1 + and isinstance(data.get("capability"), str) + and 16 <= len(data["capability"]) <= 256 + ): + capability = data["capability"] + else: + telemetry_error = ( + "session telemetry unavailable or authorization refused" + ) + + def contains_credential(value): + if isinstance(value, str): + return any( + secret and secret in value + for secret in (validation_token, capability) + ) + if isinstance(value, dict): + return any( + contains_credential(key) or contains_credential(child) + for key, child in value.items() + ) + if isinstance(value, list): + return any(contains_credential(child) for child in value) + return False + def exchange(operation: str, data: dict | None = None) -> dict: nonlocal phase, trace_bytes phase = operation @@ -174,11 +228,23 @@ def exchange(operation: str, data: dict | None = None) -> dict: raw = connection.recv(timeout=remaining()) if not isinstance(raw, str): raise ValueError("binary response is not the JSON protocol") + if len(raw.encode("utf-8")) > MAX_MESSAGE_BYTES: + raise ValueError("response exceeds size bound") trace_bytes += len(raw.encode("utf-8")) + len( request_json.encode("utf-8") ) if trace_bytes > MAX_TRACE_BYTES: raise ValueError("trace exceeds total size bound") + # Subjects can echo the authorization into arbitrary JSON fields. + # Check decoded values as well as raw text (which may be malformed). + if contains_credential(raw): + raise ValueError("response contains validation credentials") + try: + parsed = json.loads(raw) + except (ValueError, RecursionError): + parsed = None + if contains_credential(parsed): + raise ValueError("response contains validation credentials") exchanges.append( WireExchange( operation=operation, @@ -205,6 +271,29 @@ def exchange(operation: str, data: dict | None = None) -> dict: break observation = exchange("step", action) exchange("state") + if capability: + phase = "validation_read" + try: + response = telemetry_request( + phase, {"schema_version": 1, "capability": capability} + ) + if response.get("type") != "validation" or not isinstance( + response.get("data"), dict + ): + raise ValueError("invalid telemetry envelope") + if contains_credential(response["data"]): + raise ValueError("telemetry contains validation credentials") + snapshot = json.dumps( + response["data"], + allow_nan=False, + ensure_ascii=False, + separators=(",", ":"), + ) + if len(snapshot.encode("utf-8")) > MAX_TRACE_BYTES: + raise ValueError("telemetry exceeds size bound") + telemetry_json = snapshot + except Exception as exc: + telemetry_error = f"session telemetry failed ({type(exc).__name__})" complete = True finally: # Teardown is best effort and cannot replace an in-episode failure @@ -223,7 +312,10 @@ def exchange(operation: str, data: dict | None = None) -> dict: except (Exception, KeyboardInterrupt): _abort_transport(connection) return RuntimeEvidence( - exchanges=tuple(exchanges), observation_schema_json=schema_json + exchanges=tuple(exchanges), + observation_schema_json=schema_json, + telemetry_json=telemetry_json, + telemetry_error=telemetry_error, ) except KeyboardInterrupt: raise RuntimeCollectionInterrupted( @@ -232,6 +324,8 @@ def exchange(operation: str, data: dict | None = None) -> dict: observation_schema_json=schema_json, failure_phase=phase, failure_reason=f"{phase} failed (KeyboardInterrupt)", + telemetry_json=telemetry_json, + telemetry_error=telemetry_error, ) ) from None except Exception as exc: @@ -241,4 +335,6 @@ def exchange(operation: str, data: dict | None = None) -> dict: observation_schema_json=schema_json, failure_phase=phase, failure_reason=f"{phase} failed ({type(exc).__name__})", + telemetry_json=telemetry_json, + telemetry_error=telemetry_error, ) diff --git a/src/openenv/validation/runtime/contracts.py b/src/openenv/validation/runtime/contracts.py index 3d84e7be68..9df4a3eb1b 100644 --- a/src/openenv/validation/runtime/contracts.py +++ b/src/openenv/validation/runtime/contracts.py @@ -265,9 +265,15 @@ class RuntimeEvidence: Collector phase that failed; a truncated transcript cannot pass silently. failure_reason (`str`, *optional*): Credential-safe explanation of the collection failure. + telemetry_json (`str`, *optional*): + Subject-emitted session snapshot, independent of the wire transcript. + telemetry_error (`str`, *optional*): + Bounded telemetry failure without invalidating completed wire evidence. """ exchanges: tuple[WireExchange, ...] = () observation_schema_json: str | None = None failure_phase: str | None = None failure_reason: str | None = None + telemetry_json: str | None = None + telemetry_error: str | None = None diff --git a/tests/fixtures/validation/runtime/served_probe/app.py b/tests/fixtures/validation/runtime/served_probe/app.py index 91999cb91c..edaf208448 100644 --- a/tests/fixtures/validation/runtime/served_probe/app.py +++ b/tests/fixtures/validation/runtime/served_probe/app.py @@ -9,6 +9,7 @@ from openenv.core.env_server.http_server import create_app from openenv.core.env_server.interfaces import Environment from openenv.core.env_server.types import Action, Observation, State +from openenv.core.rubrics import Rubric, WeightedSum from pydantic import Field @@ -20,6 +21,14 @@ class ProbeObservation(Observation): counter: int = Field(strict=True) +class CounterRubric(Rubric): + def forward(self, action, observation): + return float(observation.counter >= 2) + + def validation_config(self): + return {"threshold": 2} + + class ProbeEnvironment(Environment): SUPPORTS_CONCURRENT_SESSIONS = True @@ -27,6 +36,7 @@ def __init__(self): super().__init__() self._state = State(episode_id="uninitialized", step_count=0) self.counter = 0 + self.rubric = WeightedSum([CounterRubric(), CounterRubric()], [0.5, 0.5]) def reset(self, seed=None, episode_id=None, **kwargs): self.counter = 0 @@ -36,11 +46,13 @@ def reset(self, seed=None, episode_id=None, **kwargs): def step(self, action, timeout_s=None, **kwargs): self.counter += action.increment self._state.step_count += 1 - return ProbeObservation( + observation = ProbeObservation( counter=self.counter, reward=float(self.counter >= 2), done=self._state.step_count >= 2, ) + observation.reward = self.rubric(action, observation) + return observation @property def state(self): diff --git a/tests/test_validation/integration/test_runtime_process.py b/tests/test_validation/integration/test_runtime_process.py new file mode 100644 index 0000000000..bf6b68c31a --- /dev/null +++ b/tests/test_validation/integration/test_runtime_process.py @@ -0,0 +1,219 @@ +"""Real HTTP/WebSocket evidence with installed OpenEnv and a test-only process. + +This covers protocol and grading on CPU Jobs. It does not establish Docker, +resource-limit, network-isolation or fresh-container acceptance. +""" + +import hashlib +import importlib.metadata +import json +import os +import shutil +import signal +import socket +import subprocess +import sys +import sysconfig +import time +from pathlib import Path + +import httpx +import pytest +from openenv.validation.providers import StartupError, UnsupportedCapability +from openenv.validation.runner import run_validation +from openenv.validation.types import CheckStatus, Level, ProviderCapability + +FIXTURE = Path(__file__).parents[2] / "fixtures/validation/runtime/served_probe" +SERVER = """ +import importlib.util, os, sys, uvicorn +spec = importlib.util.spec_from_file_location('process_probe', sys.argv[1]) +module = importlib.util.module_from_spec(spec) +spec.loader.exec_module(module) +uvicorn.run(module.make_app(os.environ['VALIDATION_FAULT']), + fd=int(sys.argv[2]), log_level='warning') +""" + + +class ProcessSubject: + def __init__(self, process, port, log_path, record_sha256): + self.process = process + self.port = port + self.log_path = log_path + self.record_sha256 = record_sha256 + self.base_url = f"http://127.0.0.1:{port}" + + def inspect(self): + return { + "test_only": True, + "isolation": "process", + "pid": self.process.pid, + "installed_record_sha256": self.record_sha256, + "container_build_exercised": False, + "resource_limits_enforced": False, + "network_policy_enforced": False, + } + + def logs(self, max_bytes=65536): + with self.log_path.open("rb") as stream: + stream.seek(max(0, self.log_path.stat().st_size - max_bytes)) + return stream.read(max_bytes).decode("utf-8", "replace") + + def exec(self, argv, timeout_s): + raise UnsupportedCapability("test process provider has no sandbox exec") + + def stop(self): + if self.process.poll() is None: + os.killpg(self.process.pid, signal.SIGTERM) + try: + self.process.wait(timeout=3) + except subprocess.TimeoutExpired: + os.killpg(self.process.pid, signal.SIGKILL) + self.process.wait(timeout=3) + assert self.process.poll() is not None + with socket.socket() as probe: + probe.settimeout(1) + assert probe.connect_ex(("127.0.0.1", self.port)) != 0 + + +class ProcessProvider: + """Substitute only process launch; all wire collection and grading are real. + + IMAGE_BUILD satisfies the runner's lifecycle seam in this test only. The + reference identifies installed wheel metadata, never a purported Docker image. + """ + + name = "test-process" + capabilities = frozenset({ProviderCapability.IMAGE_BUILD}) + supported_network_modes = frozenset({"public"}) + + def __init__(self, work, mode="good"): + self.work = work + self.work.mkdir(parents=True) + self.mode = mode + self.subjects = [] + installed = next( + item + for item in importlib.metadata.distributions( + path=[sysconfig.get_path("purelib")] + ) + if item.metadata["Name"] == "openenv" + ) + self.record_sha256 = hashlib.sha256( + installed.read_text("RECORD").encode() + ).hexdigest() + + def build(self, root, execution): + self.package = self.work / "fixture" + shutil.copytree( + root, self.package, ignore=shutil.ignore_patterns("__pycache__") + ) + return "sha256:" + self.record_sha256 + + def start(self, spec): + assert spec.network.mode == "public" and spec.resources.gpus == 0 + log_path = self.work / f"server-{len(self.subjects)}.log" + with socket.socket() as listener, log_path.open("wb") as log: + listener.bind(("127.0.0.1", 0)) + port = listener.getsockname()[1] + process = subprocess.Popen( + [ + sys.executable, + "-I", + "-c", + SERVER, + str(self.package / "app.py"), + str(listener.fileno()), + ], + stdin=subprocess.DEVNULL, + stdout=log, + stderr=subprocess.STDOUT, + env={ + "PATH": os.defpath, + "GRADIO_ANALYTICS_ENABLED": "False", + "VALIDATION_FAULT": self.mode, + **spec.env_vars, + }, + pass_fds=(listener.fileno(),), + start_new_session=True, + ) + subject = ProcessSubject(process, port, log_path, self.record_sha256) + self.subjects.append(subject) + deadline = time.monotonic() + min(spec.startup_timeout_s, 10) + try: + with httpx.Client(trust_env=False, timeout=0.2) as client: + while process.poll() is None and time.monotonic() < deadline: + try: + if client.get(subject.base_url + "/health").status_code == 200: + return subject + except httpx.HTTPError: + pass + time.sleep(0.02) + raise StartupError("test process failed readiness") + except BaseException: + subject.stop() + raise + + +@pytest.mark.parametrize( + "mode,failed_check", + [ + ("good", None), + ("bad_reward", "runtime.reward_well_formed"), + ("bad_observation", "runtime.observation_schema"), + ("missing_done", "runtime.observation_schema"), + ("bad_state", "runtime.state_contract"), + ("startup_failure", "runtime.startup"), + ], +) +def test_installed_server_collector_and_graders_over_loopback( + tmp_path, mode, failed_check +): + artifacts = ( + Path(os.environ.get("OPENENV_VALIDATION_ARTIFACTS", tmp_path)) + / "process" + / mode + ) + provider = ProcessProvider(artifacts / "subject", mode) + try: + report = run_validation( + FIXTURE, + max_level=Level.RUNTIME, + provider=provider, + artifacts_dir=artifacts / "report", + ) + finally: + for subject in provider.subjects: + subject.stop() + checks = {result.check_id: result for result in report.results} + assert checks["static.manifest"].status is CheckStatus.PASS + if failed_check: + assert checks[failed_check].status is CheckStatus.FAIL + else: + for check in ( + "startup", + "reward_well_formed", + "observation_schema", + "state_contract", + ): + assert checks[f"runtime.{check}"].status is CheckStatus.PASS + if mode != "startup_failure": + trace = json.loads((artifacts / "report/collector-trace.json").read_text()) + assert sum(row["operation"] == "step" for row in trace) == 2 + manifest = json.loads((artifacts / "report/run-manifest.json").read_text()) + assert manifest["provider"]["isolation"] == "process" + assert manifest["provider"]["container_build_exercised"] is False + telemetry_path = artifacts / "report/session-telemetry.json" + telemetry = json.loads(telemetry_path.read_text()) + assert telemetry["seed"]["accepted"] is True + assert len(telemetry["trajectory"]["records"]) == len(trace) + assert len(telemetry["attribution"]) == 2 + assert telemetry["trajectory"]["complete"] is True + for line in (artifacts / "report/SHA256SUMS").read_text().splitlines(): + checksum, name = line.split(" ", 1) + assert ( + hashlib.sha256((artifacts / "report" / name).read_bytes()).hexdigest() + == checksum + ) + assert provider.subjects and all( + subject.process.poll() is not None for subject in provider.subjects + ) diff --git a/tests/test_validation/integration/test_session_telemetry_protocol.py b/tests/test_validation/integration/test_session_telemetry_protocol.py new file mode 100644 index 0000000000..d4586a2665 --- /dev/null +++ b/tests/test_validation/integration/test_session_telemetry_protocol.py @@ -0,0 +1,275 @@ +"""Exercise the real replay WebSocket and its separate production MCP boundary.""" + +import json + +import pytest +from fastapi import FastAPI +from fastapi.testclient import TestClient +from fastmcp import FastMCP +from openenv.core.env_server.http_server import HTTPEnvServer +from openenv.core.env_server.interfaces import Environment +from openenv.core.env_server.types import Action, Observation, State +from openenv.core.rubrics import Rubric, WeightedSum + +TOKEN = "per-run-validation-test-token-000000000000" + + +class ValueAction(Action): + value: float + + +class ValueScore(Rubric): + def forward(self, action, observation): + return action.value + + def validation_config(self): + return {} + + +class SessionEnv(Environment): + SUPPORTS_CONCURRENT_SESSIONS = True + + def __init__(self): + super().__init__(rubric=WeightedSum([ValueScore(), ValueScore()], [0.25, 0.75])) + self.count = 0 + self.seed = None + self.mcp_server = FastMCP("telemetry-test") + + def reset(self, seed=None): + self.seed, self.count = seed, 0 + return Observation(reward=0.0, metadata={"seed": seed}) + + def step(self, action): + self.count += 1 + observation = Observation(metadata={"count": self.count}) + observation.reward = self.rubric(action, observation) + return observation + + @property + def state(self): + return State(episode_id="same-session", step_count=self.count) + + +class DropsSeed(SessionEnv): + def reset(self): + return super().reset() + + +class KwargSeed(SessionEnv): + def reset(self, **kwargs): + return super().reset(seed=kwargs.get("seed")) + + +class AsyncSeed(SessionEnv): + async def reset_async(self, seed=None): + return super().reset(seed=seed) + + +class AsyncDropsSeed(SessionEnv): + async def reset_async(self): + return super().reset() + + +def app_for(monkeypatch, env=SessionEnv, *, enabled=True, mode="simulation"): + if enabled: + monkeypatch.setenv("OPENENV_VALIDATION_TOKEN", TOKEN) + else: + monkeypatch.delenv("OPENENV_VALIDATION_TOKEN", raising=False) + app = FastAPI() + HTTPEnvServer(env, ValueAction, Observation, max_concurrent_envs=4).register_routes( + app, mode=mode + ) + return app + + +def exchange(ws, request): + ws.send_json(request) + return ws.receive_json() + + +def authorize(ws): + response = exchange( + ws, {"type": "validation_open", "data": {"schema_version": 1, "token": TOKEN}} + ) + assert response["type"] == "validation_open", response + return response["data"]["capability"] + + +def read(ws, capability): + return exchange( + ws, + { + "type": "validation_read", + "data": {"schema_version": 1, "capability": capability}, + }, + ) + + +@pytest.mark.parametrize( + "env,accepted", + [ + (SessionEnv, True), + (DropsSeed, False), + (AsyncSeed, True), + (AsyncDropsSeed, False), + (KwargSeed, True), + ], +) +def test_same_session_seed_scores_and_subject_record(monkeypatch, env, accepted): + with TestClient(app_for(monkeypatch, env)) as client: + with client.websocket_connect("/ws") as ws: + capability = authorize(ws) + requests = [ + {"type": "reset", "data": {"seed": 42}}, + {"type": "state"}, + {"type": "step", "data": {"value": 0.6}}, + {"type": "state"}, + ] + responses = [exchange(ws, request) for request in requests] + snapshot = read(ws, capability)["data"] + assert snapshot["seed"] == { + "requested": True, + "value": 42, + "accepted": accepted, + } + assert responses[-1]["data"]["step_count"] == 1 + assert responses[0]["data"]["observation"]["metadata"]["seed"] == ( + 42 if accepted else None + ) + assert snapshot["trajectory"]["complete"] is True + assert snapshot["trajectory"]["source"] == "openenv-server" + assert snapshot["trajectory"]["records"] == [ + {"operation": req["type"], "request": req, "response": resp} + for req, resp in zip(requests, responses) + ] + assert len(snapshot["attribution"]) == 1 + assert snapshot["attribution"][0]["step_index"] == 0 + assert all(node["evaluated"] for node in snapshot["rubric"]) + assert snapshot["rubric"][0]["score"] == pytest.approx(0.6) + # Editing the collector's copy cannot modify the subject's retained record. + responses[-1]["data"]["step_count"] = 999 + assert ( + read(ws, capability)["data"]["trajectory"]["records"][-1]["response"][ + "data" + ]["step_count"] + == 1 + ) + assert TOKEN not in json.dumps(snapshot) + assert capability not in json.dumps(snapshot) + + +@pytest.mark.parametrize("enabled,mode", [(False, "simulation"), (True, "production")]) +def test_validation_is_opt_in_and_simulation_only(monkeypatch, enabled, mode): + with TestClient(app_for(monkeypatch, enabled=enabled, mode=mode)) as client: + with client.websocket_connect("/ws") as ws: + denied = exchange( + ws, + { + "type": "validation_open", + "data": {"schema_version": 1, "token": TOKEN}, + }, + ) + assert denied["type"] == "error" + assert TOKEN not in json.dumps(denied) + # Ordinary replay clients do not need to know about telemetry. + assert ( + exchange(ws, {"type": "reset", "data": {"seed": 4}})["type"] + == "observation" + ) + + +def test_missing_wrong_malformed_cross_socket_and_expired_capabilities(monkeypatch): + with TestClient(app_for(monkeypatch)) as client: + with ( + client.websocket_connect("/ws") as first, + client.websocket_connect("/ws") as second, + ): + assert read(first, "x" * 32)["type"] == "error" + for data in ( + {}, + {"schema_version": 1, "token": "wrong-" * 8}, + {"schema_version": 999, "token": TOKEN}, + ): + denied = exchange(first, {"type": "validation_open", "data": data}) + assert denied["type"] == "error" + assert TOKEN not in json.dumps(denied) + assert "wrong-" not in json.dumps(denied) + first_cap, second_cap = authorize(first), authorize(second) + assert first_cap != second_cap + assert read(second, first_cap)["type"] == "error" + assert read(first, second_cap)["type"] == "error" + assert read(first, first_cap)["type"] == "validation" + with client.websocket_connect("/ws") as fresh: + authorize(fresh) + assert read(fresh, first_cap)["type"] == "error" + + +def test_late_open_cannot_discard_prior_replay_operations(monkeypatch): + with ( + TestClient(app_for(monkeypatch)) as client, + client.websocket_connect("/ws") as ws, + ): + exchange(ws, {"type": "reset", "data": {"seed": 0}}) + assert ( + exchange( + ws, + { + "type": "validation_open", + "data": {"schema_version": 1, "token": TOKEN}, + }, + )["type"] + == "error" + ) + + +def test_production_mcp_has_no_telemetry_or_reset_tools(monkeypatch): + with TestClient(app_for(monkeypatch, mode="production")) as client: + assert client.post("/reset", json={}).status_code == 404 + with client.websocket_connect("/mcp") as ws: + listed = exchange( + ws, {"jsonrpc": "2.0", "id": 1, "method": "tools/list", "params": {}} + ) + assert listed["result"]["tools"] == [] + for name in ("reset", "validation_open", "validation_read"): + result = exchange( + ws, + { + "jsonrpc": "2.0", + "id": 2, + "method": "tools/call", + "params": {"name": name, "arguments": {}}, + }, + ) + assert "error" in result + + +@pytest.mark.parametrize("broken", [False, True]) +def test_absent_or_uninspectable_rubric_does_not_hide_subject_record( + monkeypatch, broken +): + class NoRubricEnv(SessionEnv): + def __init__(self): + super().__init__() + if broken: + # Exact stock containers have known config; use a custom leaf. + self.rubric = ValueScore() + self.rubric.validation_config = lambda: {"private": object()} + else: + self.rubric = None + + def step(self, action): + self.count += 1 + return Observation(reward=0.5) + + with ( + TestClient(app_for(monkeypatch, NoRubricEnv)) as client, + client.websocket_connect("/ws") as ws, + ): + capability = authorize(ws) + exchange(ws, {"type": "reset", "data": {"seed": 1}}) + exchange(ws, {"type": "step", "data": {"value": 0.5}}) + snapshot = read(ws, capability)["data"] + assert snapshot["trajectory"]["complete"] is True + assert len(snapshot["trajectory"]["records"]) == 2 + assert snapshot["rubric"] == [] + assert bool(snapshot["rubric_error"]) is broken diff --git a/tests/test_validation/test_runtime_artifacts.py b/tests/test_validation/test_runtime_artifacts.py index 3921447f8a..490d28f6b0 100644 --- a/tests/test_validation/test_runtime_artifacts.py +++ b/tests/test_validation/test_runtime_artifacts.py @@ -2,6 +2,7 @@ import json from dataclasses import replace +import pytest from conftest import load_fixture_manifest from openenv.validation.graders import Subject from openenv.validation.graders.runtime import ( @@ -222,3 +223,16 @@ def test_malformed_wire_omission_is_visible_in_metadata(tmp_path): assert metadata["omitted_trace_fields"] == [ {"exchange_index": 0, "field": "response_json"} ] + + +@pytest.mark.parametrize( + "telemetry", ['{"rubric":[{"config":{"api_key":"private-value"}}]}', "{broken"] +) +def test_telemetry_redaction_or_omission_marks_bundle_modified(tmp_path, telemetry): + original = replace(measured(), telemetry_json=telemetry) + write_runtime_bundle(tmp_path, report(), evidence=original) + metadata = json.loads((tmp_path / "collector-evidence.json").read_text()) + assert metadata["redacted"] is True + assert "private-value" not in "".join( + path.read_text() for path in tmp_path.iterdir() + ) diff --git a/tests/test_validation/test_runtime_collector.py b/tests/test_validation/test_runtime_collector.py index 0741e2e8a5..c6ab1b36bc 100644 --- a/tests/test_validation/test_runtime_collector.py +++ b/tests/test_validation/test_runtime_collector.py @@ -105,3 +105,15 @@ def test_uncompressed_schema_still_obeys_total_byte_budget(monkeypatch, plan): assert evidence.failure_phase == "schema" assert evidence.failure_reason == "schema failed (ValueError)" assert evidence.observation_schema_json is None + + +def test_schema_cannot_persist_validation_credential(monkeypatch, plan): + token = "validation-secret-value-" * 2 + stream = TrackedStream(json.dumps({"observation": {"description": token}}).encode()) + schema_transport(monkeypatch, stream) + evidence = collector.collect_runtime_evidence( + "http://127.0.0.1:8000", plan, episode_timeout_s=2, validation_token=token + ) + assert evidence.observation_schema_json is None + assert evidence.failure_phase == "schema" + assert token not in str(evidence) diff --git a/tests/test_validation/test_runtime_telemetry_collector.py b/tests/test_validation/test_runtime_telemetry_collector.py new file mode 100644 index 0000000000..754c8792c9 --- /dev/null +++ b/tests/test_validation/test_runtime_telemetry_collector.py @@ -0,0 +1,111 @@ +"""Telemetry cannot expand its byte budget or retain replay credentials.""" + +import json + +import httpx +import pytest +from openenv.validation.runtime import collector +from openenv.validation.runtime.contracts import RuntimePlan + +TOKEN = "run-authorization-" + "x" * 32 +CAPABILITY = "socket-capability-" + "y" * 32 + + +def collect(monkeypatch, *, snapshot=None, leaked_response=None, escaped=False): + original = httpx.Client + monkeypatch.setattr( + collector.httpx, + "Client", + lambda **kwargs: original( + transport=httpx.MockTransport( + lambda request: httpx.Response(200, json={"observation": {}}) + ), + **kwargs, + ), + ) + + class Connection: + def send(self, raw): + self.request = json.loads(raw) + + def recv(self, timeout): + operation = self.request["type"] + if operation == "validation_open": + response = { + "type": operation, + "data": {"schema_version": 1, "capability": CAPABILITY}, + } + elif operation == "validation_read": + response = { + "type": "validation", + "data": snapshot or {"schema_version": 1}, + } + elif operation == "state": + response = { + "type": "state", + "data": {"episode_id": "test", "step_count": 0}, + } + else: + response = { + "type": "observation", + "data": { + "observation": {"message": leaked_response}, + "done": False, + "reward": 0.0, + }, + } + raw = json.dumps(response, ensure_ascii=False, separators=(",", ":")) + if escaped and leaked_response: + raw = raw.replace( + leaked_response, + "".join(f"\\u{ord(c):04x}" for c in leaked_response), + ) + return raw + + def close(self): + pass + + monkeypatch.setattr(collector, "connect", lambda *args, **kwargs: Connection()) + plan = RuntimePlan.model_validate( + { + "plan_schema_version": "1", + "reset": {"episode_id": "test", "seed": 1}, + "actions": [{"increment": 1}], + } + ) + return collector.collect_runtime_evidence( + "http://127.0.0.1:8000", plan, episode_timeout_s=2, validation_token=TOKEN + ) + + +def test_utf8_telemetry_stays_inside_received_budget(monkeypatch): + monkeypatch.setattr(collector, "MAX_TRACE_BYTES", 2048) + result = collect(monkeypatch, snapshot={"schema_version": 1, "text": "😀" * 400}) + assert result.telemetry_error is None + assert len(result.telemetry_json.encode()) <= 2048 + assert "😀" in result.telemetry_json + + +@pytest.mark.parametrize("credential", [TOKEN, CAPABILITY]) +def test_telemetry_cannot_persist_credentials_under_arbitrary_keys( + monkeypatch, credential +): + result = collect(monkeypatch, snapshot={"schema_version": 1, "debug": [credential]}) + assert result.telemetry_json is None + assert result.telemetry_error == "session telemetry failed (ValueError)" + assert result.failure_reason is None + assert len(result.exchanges) == 4 + assert credential not in repr(result) + + +@pytest.mark.parametrize( + "credential,escaped", + [(TOKEN, False), (CAPABILITY, False), (TOKEN, True), (CAPABILITY, True)], +) +def test_normal_wire_cannot_persist_plain_or_escaped_credentials( + monkeypatch, credential, escaped +): + result = collect(monkeypatch, leaked_response=credential, escaped=escaped) + assert result.exchanges == () + assert result.failure_reason == "reset failed (ValueError)" + assert credential not in repr(result) diff --git a/tests/test_validation/test_session_telemetry.py b/tests/test_validation/test_session_telemetry.py new file mode 100644 index 0000000000..c3a313e8f5 --- /dev/null +++ b/tests/test_validation/test_session_telemetry.py @@ -0,0 +1,102 @@ +"""Evidence remains detached, bounded, and fresh across rubric evaluation paths.""" + +import asyncio + +import pytest +from openenv.core.env_server import session_telemetry +from openenv.core.env_server.session_telemetry import ( + rubric_counts, + rubric_snapshot, + SessionTelemetry, +) +from openenv.core.rubrics import Gate, Rubric, Sequential, WeightedSum + + +class PublicScore(Rubric): + def forward(self, action, observation): + return action + + def validation_config(self): + return {} + + +class AsyncScore(PublicScore): + async def forward(self, action, observation): + return action + + +@pytest.mark.parametrize("score_cls", [PublicScore, AsyncScore]) +def test_gating_excludes_stale_scores_in_sync_and_async_paths(score_cls): + rubric = Sequential(Gate(score_cls(), threshold=0.5), score_cls()) + + def score(value): + result = rubric(value, None) + return asyncio.run(result) if asyncio.iscoroutine(result) else result + + score(1.0) + before = rubric_counts(rubric) + assert score(0.2) == 0.0 + nodes = {node.name: node for node in rubric_snapshot(rubric, before)} + assert nodes["root"].evaluated and nodes["root"].score == 0.0 + assert nodes["root.rubric_0"].aggregation == "gate" + assert nodes["root.rubric_0"].config == {"threshold": 0.5} + assert nodes["root.rubric_0.rubric"].score == 0.2 + assert nodes["root.rubric_1"].evaluated is False + assert nodes["root.rubric_1"].score is None + assert rubric.rubric_1.last_score == 1.0 + + +def test_weighted_semantics_and_private_configuration_are_explicit(): + class PrivateScore(Rubric): + def __init__(self): + super().__init__() + self.api_key = "must-not-appear" + + def forward(self, action, observation): + return 0.5 + + def state_dict(self): + return {"secret": self.api_key} + + rubric = WeightedSum([PublicScore(), PrivateScore()], [0.2, 0.8]) + before = rubric_counts(rubric) + assert rubric(1.0, None) == pytest.approx(0.6) + nodes = rubric_snapshot(rubric, before) + assert nodes[0].aggregation == "weighted_sum" + assert nodes[0].config == {"weights": [0.2, 0.8]} + assert nodes[0].children == ["root.rubric_0", "root.rubric_1"] + assert nodes[2].config_available is False + assert "must-not-appear" not in str([node.model_dump() for node in nodes]) + + +def test_records_are_detached_and_action_and_byte_limits_are_explicit(monkeypatch): + subject = SessionTelemetry() + response = {"type": "observation", "data": {"observation": {"counter": 1}}} + subject.append("step", {"type": "step", "data": {}}, response) + response["data"]["observation"]["counter"] = 999 + assert subject.snapshot.trajectory.records[0].response["data"]["observation"] == { + "counter": 1 + } + for _ in range(100): + subject.append("step", {"type": "step", "data": {}}, response) + assert len(subject.snapshot.trajectory.records) == 100 + assert subject.snapshot.trajectory.complete is False + assert "budget" in subject.snapshot.trajectory.reason + + monkeypatch.setattr(session_telemetry, "MAX_TELEMETRY_BYTES", 4200) + subject = SessionTelemetry() + subject.append( + "state", {"type": "state"}, {"type": "state", "data": {"x": "x" * 200}} + ) + assert subject.snapshot.trajectory.complete is False + assert subject.snapshot.trajectory.records == [] + + +def test_non_json_or_cyclic_rubrics_cannot_claim_complete_evidence(): + subject = SessionTelemetry() + subject.append("step", {"type": "step", "data": {}}, {"reward": float("nan")}) + assert subject.snapshot.trajectory.complete is False + rubric = PublicScore() + rubric.child = rubric + with pytest.raises(ValueError, match="cyclic"): + rubric_snapshot(rubric) diff --git a/tests/validation_runtime/README.md b/tests/validation_runtime/README.md index 75ddafb443..4b27945737 100644 --- a/tests/validation_runtime/README.md +++ b/tests/validation_runtime/README.md @@ -17,6 +17,13 @@ failure. The reference job uses Linux x86-64; Docker Desktop arm64 uses the same recipe but records its different platform. No host Python code from the subject is imported by the validator. +The protocol suite also exercises the installed wheel over real loopback HTTP +and WebSocket connections using a test-only process provider. It verifies +authorized session telemetry, full collection/grading/reporting, and process +cleanup on CPU-only Hugging Face Jobs. Its evidence explicitly records process +isolation; it does not qualify Docker, resource or network isolation. No Hub or +GitHub credentials are needed by those tests. + The Docker suite snapshots the current source, builds its exact OpenEnv wheel, and installs that wheel into this dedicated non-editable test environment. It downloads only binary dependencies selected from the committed lock for the diff --git a/tests/validation_runtime/acceptance.json b/tests/validation_runtime/acceptance.json index d8a3d1d1c3..75ad404610 100644 --- a/tests/validation_runtime/acceptance.json +++ b/tests/validation_runtime/acceptance.json @@ -7,7 +7,25 @@ "tests.test_validation.integration.test_served_probe::test_fault_is_visible_in_raw_wire_response[bad_observation]", "tests.test_validation.integration.test_served_probe::test_fault_is_visible_in_raw_wire_response[missing_done]", "tests.test_validation.integration.test_served_probe::test_fault_is_visible_in_raw_wire_response[bad_state]", - "tests.test_validation.integration.test_served_probe::test_failed_start_is_explicit" + "tests.test_validation.integration.test_served_probe::test_failed_start_is_explicit", + "tests.test_validation.integration.test_session_telemetry_protocol::test_same_session_seed_scores_and_subject_record[SessionEnv-True]", + "tests.test_validation.integration.test_session_telemetry_protocol::test_same_session_seed_scores_and_subject_record[DropsSeed-False]", + "tests.test_validation.integration.test_session_telemetry_protocol::test_same_session_seed_scores_and_subject_record[AsyncSeed-True]", + "tests.test_validation.integration.test_session_telemetry_protocol::test_same_session_seed_scores_and_subject_record[AsyncDropsSeed-False]", + "tests.test_validation.integration.test_session_telemetry_protocol::test_same_session_seed_scores_and_subject_record[KwargSeed-True]", + "tests.test_validation.integration.test_session_telemetry_protocol::test_validation_is_opt_in_and_simulation_only[False-simulation]", + "tests.test_validation.integration.test_session_telemetry_protocol::test_validation_is_opt_in_and_simulation_only[True-production]", + "tests.test_validation.integration.test_session_telemetry_protocol::test_missing_wrong_malformed_cross_socket_and_expired_capabilities", + "tests.test_validation.integration.test_session_telemetry_protocol::test_late_open_cannot_discard_prior_replay_operations", + "tests.test_validation.integration.test_session_telemetry_protocol::test_production_mcp_has_no_telemetry_or_reset_tools", + "tests.test_validation.integration.test_session_telemetry_protocol::test_absent_or_uninspectable_rubric_does_not_hide_subject_record[False]", + "tests.test_validation.integration.test_session_telemetry_protocol::test_absent_or_uninspectable_rubric_does_not_hide_subject_record[True]", + "tests.test_validation.integration.test_runtime_process::test_installed_server_collector_and_graders_over_loopback[good-None]", + "tests.test_validation.integration.test_runtime_process::test_installed_server_collector_and_graders_over_loopback[bad_reward-runtime.reward_well_formed]", + "tests.test_validation.integration.test_runtime_process::test_installed_server_collector_and_graders_over_loopback[bad_observation-runtime.observation_schema]", + "tests.test_validation.integration.test_runtime_process::test_installed_server_collector_and_graders_over_loopback[missing_done-runtime.observation_schema]", + "tests.test_validation.integration.test_runtime_process::test_installed_server_collector_and_graders_over_loopback[bad_state-runtime.state_contract]", + "tests.test_validation.integration.test_runtime_process::test_installed_server_collector_and_graders_over_loopback[startup_failure-runtime.startup]" ], "docker": [ "tests.test_validation.integration.test_docker_lifecycle::test_docker_lifecycle_effective_limits_and_owned_cleanup", From f1575e9856aa022b7c0c89fe79e41f2d6cf2906d Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Thu, 24 Sep 2026 12:16:14 +0200 Subject: [PATCH 13/29] fix: keep process evidence free of bytecode --- tests/test_validation/integration/test_runtime_process.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_validation/integration/test_runtime_process.py b/tests/test_validation/integration/test_runtime_process.py index bf6b68c31a..efe4464a24 100644 --- a/tests/test_validation/integration/test_runtime_process.py +++ b/tests/test_validation/integration/test_runtime_process.py @@ -119,6 +119,7 @@ def start(self, spec): [ sys.executable, "-I", + "-B", "-c", SERVER, str(self.package / "app.py"), From 297b578b1fc30419cb185321dbeee7cdc8f1eaf3 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Thu, 24 Sep 2026 10:18:25 +0000 Subject: [PATCH 14/29] Fix source digest portability on Windows --- src/openenv/validation/runner.py | 3 ++- tests/test_validation/test_runner.py | 8 ++++++++ 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index a3c96e0fed..6730b88241 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -58,7 +58,8 @@ def source_digest(package_root: Path) -> str: for relative_path, path in sorted(files, key=lambda item: item[0]): digest.update(relative_path.encode()) digest.update(b"\0") - fd = os.open(path, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK) + flags = os.O_RDONLY | os.O_NOFOLLOW | getattr(os, "O_NONBLOCK", 0) + fd = os.open(path, flags) with os.fdopen(fd, "rb") as source: if not stat.S_ISREG(os.fstat(source.fileno()).st_mode): raise ValueError("validation source must contain regular files") diff --git a/tests/test_validation/test_runner.py b/tests/test_validation/test_runner.py index 961261d6e8..e07cbeeddc 100644 --- a/tests/test_validation/test_runner.py +++ b/tests/test_validation/test_runner.py @@ -1,4 +1,5 @@ import hashlib +import os import shutil import pytest @@ -89,6 +90,13 @@ def test_source_digest_uses_portable_relative_paths(tmp_path): assert source_digest(package_root) == expected +def test_source_digest_does_not_require_nonblocking_open(tmp_path, monkeypatch): + (tmp_path / "file.txt").write_bytes(b"contents") + monkeypatch.delattr(os, "O_NONBLOCK", raising=False) + + assert len(source_digest(tmp_path)) == 64 + + @pytest.mark.parametrize("failure", [ValueError, OSError]) def test_initial_source_digest_failure_is_reported(monkeypatch, failure): def fail_digest(*args): From 317983cc4b55be8fa9a8fc97e00acefd554056c0 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Thu, 24 Sep 2026 13:59:14 +0200 Subject: [PATCH 15/29] fix: preserve runtime evidence on telemetry errors --- rfcs/008-environment-auto-validation.md | 7 ++- src/openenv/validation/runtime/collector.py | 36 ++++++----- .../integration/test_runtime_process.py | 2 +- .../test_runtime_telemetry_collector.py | 60 ++++++++++++++++++- 4 files changed, 85 insertions(+), 20 deletions(-) diff --git a/rfcs/008-environment-auto-validation.md b/rfcs/008-environment-auto-validation.md index 6ad227f5f0..5f0171a868 100644 --- a/rfcs/008-environment-auto-validation.md +++ b/rfcs/008-environment-auto-validation.md @@ -197,9 +197,10 @@ configuration (model, version, params) in the manifest; the oracle check becomes bit-exact. RFC 004 rubrics are **leveraged, not required**: the contract stays spec-neutral (graders read the manifest), but for the served OpenEnv format the rubric tree is the native satisfaction path — `LLMJudge` is the in-repo `llm_judged` implementation, and the -introspectability and reward-attribution graders read `named_rubrics()` / `state_dict()` / -per-child scores. Judge pinning stays a *manifest* declaration because the rubric object does not -serialize model/version/params today. +introspectability and reward-attribution graders read `named_rubrics()`, explicit +`validation_config()` and fresh per-child scores. `state_dict()` is never serialized +as validation configuration. Judge pinning stays a *manifest* declaration because +the rubric object does not serialize model/version/params today. Tolerances, margins, and variance bounds are author-declared in the manifest, **bounded by the versioned severity policy**, and carried verbatim in reports so hubs can apply stricter ceilings. diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index 5f1095e7a2..c542855a87 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -183,22 +183,28 @@ def telemetry_request(operation, data): capability = None if validation_token: phase = "validation_open" - response = telemetry_request( - phase, {"schema_version": 1, "token": validation_token} - ) - data = response.get("data") - if ( - response.get("type") == "validation_open" - and isinstance(data, dict) - and data.get("schema_version") == 1 - and isinstance(data.get("capability"), str) - and 16 <= len(data["capability"]) <= 256 - ): - capability = data["capability"] - else: - telemetry_error = ( - "session telemetry unavailable or authorization refused" + try: + response = telemetry_request( + phase, {"schema_version": 1, "token": validation_token} ) + except (ValueError, RecursionError) as exc: + # A consumed malformed reply only invalidates optional telemetry. + # Transport failure still aborts this same-session collection. + telemetry_error = f"session telemetry failed ({type(exc).__name__})" + else: + data = response.get("data") + if ( + response.get("type") == "validation_open" + and isinstance(data, dict) + and data.get("schema_version") == 1 + and isinstance(data.get("capability"), str) + and 16 <= len(data["capability"]) <= 256 + ): + capability = data["capability"] + else: + telemetry_error = ( + "session telemetry unavailable or authorization refused" + ) def contains_credential(value): if isinstance(value, str): diff --git a/tests/test_validation/integration/test_runtime_process.py b/tests/test_validation/integration/test_runtime_process.py index efe4464a24..f7e985e50b 100644 --- a/tests/test_validation/integration/test_runtime_process.py +++ b/tests/test_validation/integration/test_runtime_process.py @@ -147,7 +147,7 @@ def start(self, spec): if client.get(subject.base_url + "/health").status_code == 200: return subject except httpx.HTTPError: - pass + pass # The process may still be starting its HTTP listener. time.sleep(0.02) raise StartupError("test process failed readiness") except BaseException: diff --git a/tests/test_validation/test_runtime_telemetry_collector.py b/tests/test_validation/test_runtime_telemetry_collector.py index 754c8792c9..df1dd57146 100644 --- a/tests/test_validation/test_runtime_telemetry_collector.py +++ b/tests/test_validation/test_runtime_telemetry_collector.py @@ -6,12 +6,21 @@ import pytest from openenv.validation.runtime import collector from openenv.validation.runtime.contracts import RuntimePlan +from websockets.exceptions import ConnectionClosedError TOKEN = "run-authorization-" + "x" * 32 CAPABILITY = "socket-capability-" + "y" * 32 -def collect(monkeypatch, *, snapshot=None, leaked_response=None, escaped=False): +def collect( + monkeypatch, + *, + snapshot=None, + leaked_response=None, + escaped=False, + opening_reply=None, + opening_error=None, +): original = httpx.Client monkeypatch.setattr( collector.httpx, @@ -31,6 +40,10 @@ def send(self, raw): def recv(self, timeout): operation = self.request["type"] if operation == "validation_open": + if opening_error is not None: + raise opening_error + if opening_reply is not None: + return opening_reply response = { "type": operation, "data": {"schema_version": 1, "capability": CAPABILITY}, @@ -109,3 +122,48 @@ def test_normal_wire_cannot_persist_plain_or_escaped_credentials( assert result.exchanges == () assert result.failure_reason == "reset failed (ValueError)" assert credential not in repr(result) + + +@pytest.mark.parametrize( + "reply,error_name", + [ + (TOKEN.encode(), "ValueError"), + ("not-json-" + TOKEN, "JSONDecodeError"), + (json.dumps([TOKEN]), "ValueError"), + ], + ids=["binary", "non-json", "list"], +) +def test_malformed_opening_reply_keeps_the_ordinary_session( + monkeypatch, reply, error_name +): + result = collect(monkeypatch, opening_reply=reply) + assert result.telemetry_error == f"session telemetry failed ({error_name})" + assert result.telemetry_json is None + assert result.failure_reason is None + assert [row.operation for row in result.exchanges] == [ + "reset", + "state", + "step", + "state", + ] + assert TOKEN not in repr(result) + + +@pytest.mark.parametrize( + "error", [ConnectionClosedError(None, None), TimeoutError(TOKEN)] +) +def test_opening_transport_failure_still_aborts_collection(monkeypatch, error): + result = collect(monkeypatch, opening_error=error) + assert result.failure_phase == "validation_open" + assert result.failure_reason == f"validation_open failed ({type(error).__name__})" + assert result.telemetry_json is None + assert not result.exchanges + assert TOKEN not in repr(result) + + +def test_opening_cancellation_is_not_downgraded_to_optional_telemetry(monkeypatch): + with pytest.raises(collector.RuntimeCollectionInterrupted) as error: + collect(monkeypatch, opening_error=KeyboardInterrupt(TOKEN)) + assert error.value.evidence.failure_phase == "validation_open" + assert not error.value.evidence.exchanges + assert TOKEN not in repr(error.value.evidence) From 237d11f5c883a2c12684465c064eedb82c49f3b0 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Thu, 1 Oct 2026 12:04:48 +0200 Subject: [PATCH 16/29] fix: drop unsafe malformed telemetry frames --- src/openenv/validation/runtime/collector.py | 3 ++ .../test_runtime_telemetry_collector.py | 28 +++++++++++++++++-- 2 files changed, 29 insertions(+), 2 deletions(-) diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index c542855a87..f0f598caba 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -248,6 +248,9 @@ def exchange(operation: str, data: dict | None = None) -> dict: try: parsed = json.loads(raw) except (ValueError, RecursionError): + # Malformed JSON cannot be checked for escaped credentials. + if validation_token: + raise parsed = None if contains_credential(parsed): raise ValueError("response contains validation credentials") diff --git a/tests/test_validation/test_runtime_telemetry_collector.py b/tests/test_validation/test_runtime_telemetry_collector.py index df1dd57146..40c6d31d8f 100644 --- a/tests/test_validation/test_runtime_telemetry_collector.py +++ b/tests/test_validation/test_runtime_telemetry_collector.py @@ -17,7 +17,9 @@ def collect( *, snapshot=None, leaked_response=None, + leak_operation="reset", escaped=False, + malformed=False, opening_reply=None, opening_error=None, ): @@ -62,7 +64,11 @@ def recv(self, timeout): response = { "type": "observation", "data": { - "observation": {"message": leaked_response}, + "observation": { + "message": leaked_response + if operation == leak_operation + else None + }, "done": False, "reward": 0.0, }, @@ -73,7 +79,7 @@ def recv(self, timeout): leaked_response, "".join(f"\\u{ord(c):04x}" for c in leaked_response), ) - return raw + return raw[:-1] if malformed and operation == leak_operation else raw def close(self): pass @@ -167,3 +173,21 @@ def test_opening_cancellation_is_not_downgraded_to_optional_telemetry(monkeypatc assert error.value.evidence.failure_phase == "validation_open" assert not error.value.evidence.exchanges assert TOKEN not in repr(error.value.evidence) + + +@pytest.mark.parametrize("credential", [TOKEN, CAPABILITY]) +@pytest.mark.parametrize("operation", ["reset", "step"]) +def test_malformed_wire_cannot_retain_escaped_credentials( + monkeypatch, credential, operation +): + result = collect( + monkeypatch, + leaked_response=credential, + leak_operation=operation, + escaped=True, + malformed=True, + ) + assert [row.operation for row in result.exchanges] == ( + [] if operation == "reset" else ["reset", "state"] + ) + assert result.failure_reason == f"{operation} failed (JSONDecodeError)" From 9cdaff041c873a751fb5b0948804174da4af975e Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Thu, 1 Oct 2026 12:05:20 +0200 Subject: [PATCH 17/29] fix: runtime validation diagnostics and deadlines --- docs/source/reference/cli.md | 14 ++++++++ src/openenv/cli/commands/validate.py | 5 ++- .../validation/graders/runtime/basic.py | 16 +++++---- src/openenv/validation/runner.py | 7 ++-- src/openenv/validation/runtime/artifacts.py | 7 ++-- src/openenv/validation/runtime/collector.py | 27 ++++++++++---- .../validation/runtime/schema_worker.py | 13 +++++-- .../validation/runtime/served_probe/app.py | 8 +++-- .../integration/test_runtime_cli.py | 5 +-- .../test_validation/test_runtime_artifacts.py | 20 +++++++++++ .../test_validation/test_runtime_execution.py | 34 +++++++++++++++--- tests/test_validation/test_runtime_grading.py | 24 +++++++++++-- .../test_validation/test_runtime_transport.py | 35 ++++++++++++++++++- tests/validation_runtime/README.md | 7 ++-- tests/validation_runtime/acceptance.json | 1 + 15 files changed, 188 insertions(+), 35 deletions(-) diff --git a/docs/source/reference/cli.md b/docs/source/reference/cli.md index 77e390719b..59a7c31d59 100644 --- a/docs/source/reference/cli.md +++ b/docs/source/reference/cli.md @@ -57,6 +57,20 @@ includes the available lower-level checks but does not claim semantic execution. `--skip-build` runs declaration checks and skips runtime execution entirely. The Docker provider currently supports CPU workloads and `public` network mode; unsupported network or GPU requirements skip runtime before building. +`--local` explicitly selects this default package mode and rejects a remote URL. +The declared `episode_timeout_s` bounds collection; a reset or judged step may use +its remaining budget. Collection failures appear once under `runtime.startup`, +with dependent contract checks skipped and the completed trace retained. + +Validation does not inherit host credentials, and the CLI currently has no secret +injection mechanism. Environments requiring a judge API key can therefore fail at +session creation. The observation check currently applies the advertised schema +to reset and step responses. Step rewards must be finite numbers, even though core +models allow null rewards; these are the current RFC 008 validation rules. + +Cleanup removes run-owned containers. Built images remain in Docker's local cache +for reuse; the report records their immutable image IDs. Remove an unwanted image +with `docker image rm ` after its validation runs have finished. Runtime reports use schema version 2 and severity policy v2. Static reports for v1 manifests retain schema version 1 and policy v1. An explicit v1 policy with a diff --git a/src/openenv/cli/commands/validate.py b/src/openenv/cli/commands/validate.py index 55286002f8..b041653b48 100644 --- a/src/openenv/cli/commands/validate.py +++ b/src/openenv/cli/commands/validate.py @@ -120,7 +120,10 @@ def validate( ] = False, local: Annotated[ bool, - typer.Option("--local", help="Use Docker-local runtime validation explicitly"), + typer.Option( + "--local", + help="Explicitly select the default local-package mode (incompatible with --url)", + ), ] = False, policy_version: Annotated[ str | None, diff --git a/src/openenv/validation/graders/runtime/basic.py b/src/openenv/validation/graders/runtime/basic.py index b6700f5d73..540b362f51 100644 --- a/src/openenv/validation/graders/runtime/basic.py +++ b/src/openenv/validation/graders/runtime/basic.py @@ -8,6 +8,7 @@ from pathlib import Path from ...report import CheckResult +from ...runtime.artifacts import _redact from ...types import CheckStatus, Level @@ -44,9 +45,14 @@ def run(self, subject) -> CheckResult: evidence=["runtime evidence is unavailable"], duration_s=0, ) - if not evidence.failure_reason and not any( - row.operation == "step" for row in evidence.exchanges - ): + if evidence.failure_reason: + return CheckResult( + check_id=self.check_id, + status=CheckStatus.SKIP, + evidence=["unmet dependency: runtime.startup (collection incomplete)"], + duration_s=0, + ) + if not any(row.operation == "step" for row in evidence.exchanges): return CheckResult( check_id=self.check_id, status=CheckStatus.SKIP, @@ -54,8 +60,6 @@ def run(self, subject) -> CheckResult: duration_s=0, ) problems = [] - if evidence.failure_reason: - problems.append(evidence.failure_reason) try: problems.extend(self.check(subject, evidence)) except (ValueError, TypeError, KeyError, RecursionError, OverflowError): @@ -169,7 +173,7 @@ def check(self, subject, evidence) -> list[str]: "observation schema evaluation exceeded its resource budget" ) else: - problems.extend(json.loads(checked.stdout)) + problems.extend(_redact(json.loads(checked.stdout))) except subprocess.TimeoutExpired: problems.append("observation schema evaluation exceeded its time budget") return problems diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index 6730b88241..4db6b3eba0 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -161,8 +161,8 @@ def _runtime(subject, *, skip_build, provider): plan, episode_timeout_s=manifest.resources.episode_timeout_s, ) - # A health endpoint without a functioning protocol isn't a startup success. - if not evidence.exchanges and evidence.failure_reason: + # Protocol collection must finish before its dependent contract checks run. + if evidence.failure_reason: result = _outcome( "runtime.startup", CheckStatus.FAIL, @@ -221,8 +221,9 @@ def _runtime(subject, *, skip_build, provider): try: running.stop() cleanup["completed"] = True - except (Exception, KeyboardInterrupt): + except (Exception, KeyboardInterrupt) as exc: cleanup["completed"] = False + cleanup["reason"] = f"subject teardown failed ({type(exc).__name__})" if result is None: result = _outcome( "runtime.startup", diff --git a/src/openenv/validation/runtime/artifacts.py b/src/openenv/validation/runtime/artifacts.py index d2d8e340bb..1e2079c37d 100644 --- a/src/openenv/validation/runtime/artifacts.py +++ b/src/openenv/validation/runtime/artifacts.py @@ -9,8 +9,9 @@ from pathlib import Path _SECRET_KEY = re.compile(r"(?i)(password|secret|token|authorization|api[_-]?key)") +_TOKEN_METADATA_KEYS = {"max_tokens", "prompt_token_ids"} _TOKEN = re.compile( - r"(?:hf_[A-Za-z0-9]{8,}|(?:sk|ghp|github_pat)[-_][A-Za-z0-9_-]{8,}|(?i:bearer)\s+\S+)" + r"(? RuntimeEvidence: """ Preserve schema and reset/step/state responses without model coercion. @@ -87,8 +88,9 @@ def collect_runtime_evidence( Validated, bounded reset and action inputs. episode_timeout_s (`float`): Deadline for the complete collection, including schema retrieval. - request_timeout_s (`float`, *optional*, defaults to `5.0`): - Per-operation deadline, capped by the remaining episode budget. + request_timeout_s (`float`, *optional*): + Optional per-operation cap. By default, each operation may use the + remaining declared episode budget. Returns: [`~openenv.validation.runtime.contracts.RuntimeEvidence`]: raw evidence. @@ -98,9 +100,12 @@ def collect_runtime_evidence( schema_json = None phase = "schema" trace_bytes = 0 + server_code = None def remaining() -> float: - value = min(request_timeout_s, deadline - time.monotonic()) + value = deadline - time.monotonic() + if request_timeout_s is not None: + value = min(request_timeout_s, value) if value <= 0: raise TimeoutError("episode deadline exceeded") return value @@ -162,7 +167,7 @@ def remaining() -> float: try: def exchange(operation: str, data: dict | None = None) -> dict: - nonlocal phase, trace_bytes + nonlocal phase, trace_bytes, server_code phase = operation request = {"type": operation} if data is not None: @@ -187,6 +192,16 @@ def exchange(operation: str, data: dict | None = None) -> dict: ) ) response = json.loads(raw) + if isinstance(response, dict) and response.get("type") == "error": + data = response.get("data") + code = data.get("code") if isinstance(data, dict) else None + # Only protocol constants are safe diagnostics; never echo an + # arbitrary server message or a subject-defined error code. + if isinstance(code, str) and code in { + item.value for item in WSErrorCode + }: + server_code = code + raise ValueError("server returned an error") expected = "state" if operation == "state" else "observation" if ( not isinstance(response, dict) @@ -240,5 +255,5 @@ def exchange(operation: str, data: dict | None = None) -> dict: exchanges=tuple(exchanges), observation_schema_json=schema_json, failure_phase=phase, - failure_reason=f"{phase} failed ({type(exc).__name__})", + failure_reason=f"{phase} failed ({server_code or type(exc).__name__})", ) diff --git a/src/openenv/validation/runtime/schema_worker.py b/src/openenv/validation/runtime/schema_worker.py index 169645613f..b5ff8ef1de 100644 --- a/src/openenv/validation/runtime/schema_worker.py +++ b/src/openenv/validation/runtime/schema_worker.py @@ -39,9 +39,18 @@ def main(): observation = dict(data["observation"]) observation.update(reward=data["reward"], done=data["done"]) for error in validator.iter_errors(observation): - location = "/".join(str(x) for x in error.absolute_path)[:160] + location = "/".join(str(x) for x in error.absolute_schema_path)[:160] + missing = "" + if error.validator == "required": + names = [ + name + for name in error.validator_value + if name not in error.instance + ] + missing = "; missing properties: " + json.dumps(names[:5])[:160] problems.append( - f"exchange {row['index']}: schema mismatch at {location or '/'}" + f"exchange {row['index']}: schema mismatch at {location or '/'} " + f"({error.validator}){missing}" ) if len(problems) >= 20: break diff --git a/tests/fixtures/validation/runtime/served_probe/app.py b/tests/fixtures/validation/runtime/served_probe/app.py index 91999cb91c..586e433f36 100644 --- a/tests/fixtures/validation/runtime/served_probe/app.py +++ b/tests/fixtures/validation/runtime/served_probe/app.py @@ -61,7 +61,7 @@ async def fault_receive(): nonlocal steps message = await receive() if ( - self.mode == "hung_step" + self.mode in {"hung_step", "slow_step"} and message["type"] == "websocket.receive" and message.get("text") and json.loads(message["text"]).get("type") == "step" @@ -71,7 +71,10 @@ async def fault_receive(): # Tests signal the CLI only after the collector has completed # a real reset, first step and both corresponding state reads. os.write(1, b"OPENENV_VALIDATION_STEP_BLOCKED\n") - await asyncio.Event().wait() + if self.mode == "slow_step": + await asyncio.sleep(5.2) + else: + await asyncio.Event().wait() return message async def fault_send(message: dict[str, Any]): @@ -105,6 +108,7 @@ def make_app(mode="good"): "missing_done", "bad_state", "hung_step", + "slow_step", }: raise ValueError(f"Unknown fixture mode: {mode}") return WireFault( diff --git a/tests/test_validation/integration/test_runtime_cli.py b/tests/test_validation/integration/test_runtime_cli.py index daee42a28e..249662d400 100644 --- a/tests/test_validation/integration/test_runtime_cli.py +++ b/tests/test_validation/integration/test_runtime_cli.py @@ -289,6 +289,7 @@ def _invoke_cli( "mode,failed_check", [ ("good", None), + ("slow_step", None), ("bad_reward", "runtime.reward_well_formed"), ("bad_observation", "runtime.observation_schema"), ("missing_done", "runtime.observation_schema"), @@ -359,9 +360,9 @@ def test_cli_hung_step_times_out_with_partial_evidence(cli_context, tmp_path): assert result.returncode == 1 assert report["verdict"] == "fail" assert report["manifest"]["resources"]["episode_timeout_s"] == 3.0 - assert checks["runtime.startup"]["status"] == "pass" + assert checks["runtime.startup"]["status"] == "fail" assert all( - checks[key]["status"] == "fail" for key in IMPLEMENTED - {"runtime.startup"} + checks[key]["status"] == "skip" for key in IMPLEMENTED - {"runtime.startup"} ) _assert_partial_episode(artifacts, "TimeoutError") diff --git a/tests/test_validation/test_runtime_artifacts.py b/tests/test_validation/test_runtime_artifacts.py index 3921447f8a..fcb2cc8c36 100644 --- a/tests/test_validation/test_runtime_artifacts.py +++ b/tests/test_validation/test_runtime_artifacts.py @@ -222,3 +222,23 @@ def test_malformed_wire_omission_is_visible_in_metadata(tmp_path): assert metadata["omitted_trace_fields"] == [ {"exchange_index": 0, "field": "response_json"} ] + + +def test_redaction_preserves_token_metadata_and_filters_secret_keys(): + from openenv.validation.runtime.artifacts import _redact + + secret = "hf_abcdefghijk123456789" + source = { + "task_distribution": "task_distribution", + "max_tokens": 100, + "prompt_token_ids": [1, 2, 3], + "access_token": "opaque-credential", + secret: {"description": "public"}, + "nested": {"key": secret}, + } + filtered = _redact(source) + assert filtered["task_distribution"] == "task_distribution" + assert filtered["max_tokens"] == 100 + assert filtered["prompt_token_ids"] == [1, 2, 3] + assert filtered["access_token"] == "[REDACTED]" + assert secret not in json.dumps(filtered) diff --git a/tests/test_validation/test_runtime_execution.py b/tests/test_validation/test_runtime_execution.py index 29b0b0b7a5..ce473b9c57 100644 --- a/tests/test_validation/test_runtime_execution.py +++ b/tests/test_validation/test_runtime_execution.py @@ -330,12 +330,10 @@ def failed_stop(): result = next(r for r in report.results if r.check_id == "runtime.startup") assert result.status is CheckStatus.ERROR assert result.evidence == [ - "schema request failed" - if collection_state == "failed" - else "subject built and reached its control endpoint", + collected.failure_reason or "subject built and reached its control endpoint", "subject teardown failed", ] - if collection_state != "failed": + if collection_state == "complete": assert result.measured == { "provider": provider.name, "image_ref": "sha256:" + "a" * 64, @@ -344,6 +342,7 @@ def failed_stop(): assert json.loads((bundle / "cleanup.json").read_text()) == { "required": True, "completed": False, + "reason": f"subject teardown failed ({teardown_error.__name__})", } assert len(json.loads((bundle / "collector-trace.json").read_text())) == len( collected.exchanges @@ -452,3 +451,30 @@ def test_artifacts_encode_invalid_numbers_and_redact_secrets(package, tmp_path): payload, parse_constant=lambda x: pytest.fail(f"invalid JSON number {x}") ) assert parsed[0]["response_json"]["reward"] == {"invalid_number": "nan"} + + +@pytest.mark.parametrize("prefix_length", [0, 1, 4]) +def test_collection_failure_is_reported_once_and_dependents_skip( + package, monkeypatch, prefix_length +): + collected = replace( + measured_episode(), + exchanges=measured_episode().exchanges[:prefix_length], + failure_phase="step", + failure_reason="step failed (TimeoutError)", + ) + monkeypatch.setattr( + "openenv.validation.runner.collect_runtime_evidence", lambda *a, **k: collected + ) + report = run_validation( + package, max_level=Level.RUNTIME, provider=FakeRuntimeProvider() + ) + checks = {result.check_id: result for result in report.results} + assert report.verdict.value == "fail" + assert checks["runtime.startup"].status is CheckStatus.FAIL + assert checks["runtime.startup"].evidence == [collected.failure_reason] + for name in ("reward_well_formed", "observation_schema", "state_contract"): + assert checks[f"runtime.{name}"].status is CheckStatus.SKIP + assert checks[f"runtime.{name}"].evidence == [ + "unmet dependencies: runtime.startup" + ] diff --git a/tests/test_validation/test_runtime_grading.py b/tests/test_validation/test_runtime_grading.py index 9df9609778..b89e6f1840 100644 --- a/tests/test_validation/test_runtime_grading.py +++ b/tests/test_validation/test_runtime_grading.py @@ -169,7 +169,7 @@ def test_truncated_collection_cannot_pass_from_a_valid_prefix(tmp_path, grader): failure_reason="step failed (TimeoutError)", ), ) - assert grader().run(subject).status is CheckStatus.FAIL + assert grader().run(subject).status is CheckStatus.SKIP @pytest.mark.parametrize( @@ -369,4 +369,24 @@ def test_malformed_wire_is_a_finding_not_a_validator_crash(tmp_path, grader): failure_reason="step failed (JSONDecodeError)", ), ) - assert grader().run(subject).status is CheckStatus.FAIL + assert grader().run(subject).status is CheckStatus.SKIP + + +def test_schema_required_keyword_identifies_missing_property(tmp_path): + result = ObservationSchemaGrader().run( + subject_with(tmp_path, schema={"required": ["tool_name"]}) + ) + assert result.status is CheckStatus.FAIL + assert all("(required)" in message for message in result.evidence) + assert all( + 'missing properties: ["tool_name"]' in message for message in result.evidence + ) + + +def test_schema_diagnostics_do_not_echo_private_property_names(tmp_path): + secret = "hf_abcdefghijk123456789" + result = ObservationSchemaGrader().run( + subject_with(tmp_path, schema={"required": [secret]}) + ) + assert result.status is CheckStatus.FAIL + assert secret not in result.model_dump_json() diff --git a/tests/test_validation/test_runtime_transport.py b/tests/test_validation/test_runtime_transport.py index dc42111a58..a5545054de 100644 --- a/tests/test_validation/test_runtime_transport.py +++ b/tests/test_validation/test_runtime_transport.py @@ -94,7 +94,7 @@ def collect(monkeypatch): } ) - def run(connection, *, episode_timeout_s=2, request_timeout_s=1): + def run(connection, *, episode_timeout_s=2, request_timeout_s=None): monkeypatch.setattr(collector, "connect", lambda *args, **kwargs: connection) return collector.collect_runtime_evidence( "http://127.0.0.1:8000", @@ -253,3 +253,36 @@ def rescue(): fallback.join() sender.close() receiver.close() + + +@pytest.mark.parametrize( + "code", + ["FACTORY_ERROR", "EXECUTION_ERROR", "hf_abcdefghijk12345", {"unsafe": "payload"}], +) +def test_server_error_exposes_only_known_protocol_code(collect, code): + connection = EpisodeConnection() + connection.recv = lambda timeout: json.dumps( + {"type": "error", "data": {"code": code, "message": "private server exception"}} + ) + evidence = collect(connection) + expected = ( + code if isinstance(code, str) and code.endswith("_ERROR") else "ValueError" + ) + assert evidence.failure_reason == f"reset failed ({expected})" + assert "private server exception" not in evidence.failure_reason + assert len(evidence.exchanges) == 1 + + +def test_default_operation_deadline_uses_remaining_episode_budget(collect): + connection = EpisodeConnection() + receive = connection.recv + timeouts = [] + + def record_timeout(timeout): + timeouts.append(timeout) + return receive(timeout) + + connection.recv = record_timeout + evidence = collect(connection, episode_timeout_s=60) + assert evidence.failure_reason is None + assert all(50 < timeout <= 60 for timeout in timeouts) diff --git a/tests/validation_runtime/README.md b/tests/validation_runtime/README.md index 75ddafb443..ff6548b0a9 100644 --- a/tests/validation_runtime/README.md +++ b/tests/validation_runtime/README.md @@ -27,12 +27,13 @@ checkout with `PYTHONPATH` removed, exercising installed package data and the production OpenEnv `/ws` endpoint. Each launch uses a fresh subject. The image supports controlled `VALIDATION_FAULT` modes: `good`, `bad_reward`, -`bad_observation`, `missing_done`, `bad_state`, `hung_step`, and `startup_failure`. All fault +`bad_observation`, `missing_done`, `bad_state`, `hung_step`, `slow_step`, and `startup_failure`. All fault switches and wire corruption remain inside test assets. They share one fixture and one public runtime plan, so a defect changes one property at a time. -The Docker suite contains 13 required cases: three provider lifecycle tests, -nine CLI fault/control cases, and one real `echo_env` canary. The hung-step case +The Docker suite contains 14 required cases: three provider lifecycle tests, +ten CLI fault/control cases, and one real `echo_env` canary. The slow-step case completes a tool call taking more than five seconds within +the declared episode budget. The hung-step case checks the episode deadline; the interruption case sends SIGINT only after a container log confirms the second step has begun. Both must retain the completed reset/state/step/state prefix and remove their own containers. Each CLI case uses diff --git a/tests/validation_runtime/acceptance.json b/tests/validation_runtime/acceptance.json index d8a3d1d1c3..6ab4830800 100644 --- a/tests/validation_runtime/acceptance.json +++ b/tests/validation_runtime/acceptance.json @@ -15,6 +15,7 @@ "tests.test_validation.integration.test_docker_lifecycle::test_docker_exec_timeout_removes_process_tree", "tests.test_validation.integration.test_echo_canary::test_reference_echo_replays_real_tools_and_exposes_contract_findings", "tests.test_validation.integration.test_runtime_cli::test_cli_runtime_contract_findings[good-None]", + "tests.test_validation.integration.test_runtime_cli::test_cli_runtime_contract_findings[slow_step-None]", "tests.test_validation.integration.test_runtime_cli::test_cli_runtime_contract_findings[bad_reward-runtime.reward_well_formed]", "tests.test_validation.integration.test_runtime_cli::test_cli_runtime_contract_findings[bad_observation-runtime.observation_schema]", "tests.test_validation.integration.test_runtime_cli::test_cli_runtime_contract_findings[missing_done-runtime.observation_schema]", From aa85ab801c5e6d571e1c667383bca2b231e39970 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Thu, 1 Oct 2026 12:07:02 +0200 Subject: [PATCH 18/29] fix: hash sources portably without following swaps --- src/openenv/validation/runner.py | 20 +++++++++++----- tests/test_validation/test_runner.py | 34 ++++++++++++++++++++++++++-- 2 files changed, 46 insertions(+), 8 deletions(-) diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index 4db6b3eba0..56f2891df0 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -50,19 +50,27 @@ def source_digest(package_root: Path) -> str: relative_path = path.relative_to(package_root) if any(part in _DIGEST_EXCLUDED_DIRS for part in relative_path.parts): continue - if path.is_symlink(): + info = path.lstat() + if stat.S_ISLNK(info.st_mode): raise ValueError("validation source may not contain symbolic links") - if path.is_file(): - files.append((relative_path.as_posix(), path)) + if stat.S_ISREG(info.st_mode): + files.append((relative_path.as_posix(), path, info)) + elif not stat.S_ISDIR(info.st_mode): + raise ValueError("validation source must contain regular files") - for relative_path, path in sorted(files, key=lambda item: item[0]): + for relative_path, path, info in sorted(files, key=lambda item: item[0]): digest.update(relative_path.encode()) digest.update(b"\0") - flags = os.O_RDONLY | os.O_NOFOLLOW | getattr(os, "O_NONBLOCK", 0) + flags = ( + os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_NONBLOCK", 0) + ) fd = os.open(path, flags) with os.fdopen(fd, "rb") as source: - if not stat.S_ISREG(os.fstat(source.fileno()).st_mode): + opened = os.fstat(source.fileno()) + if not stat.S_ISREG(opened.st_mode): raise ValueError("validation source must contain regular files") + if not os.path.samestat(info, opened): + raise ValueError("validation source changed before it could be read") while chunk := source.read(1024 * 1024): digest.update(chunk) digest.update(b"\0") diff --git a/tests/test_validation/test_runner.py b/tests/test_validation/test_runner.py index e07cbeeddc..87269c4e61 100644 --- a/tests/test_validation/test_runner.py +++ b/tests/test_validation/test_runner.py @@ -90,9 +90,12 @@ def test_source_digest_uses_portable_relative_paths(tmp_path): assert source_digest(package_root) == expected -def test_source_digest_does_not_require_nonblocking_open(tmp_path, monkeypatch): +@pytest.mark.parametrize("flag", ["O_NONBLOCK", "O_NOFOLLOW"]) +def test_source_digest_does_not_require_platform_open_flags( + tmp_path, monkeypatch, flag +): (tmp_path / "file.txt").write_bytes(b"contents") - monkeypatch.delattr(os, "O_NONBLOCK", raising=False) + monkeypatch.delattr(os, flag, raising=False) assert len(source_digest(tmp_path)) == 64 @@ -119,3 +122,30 @@ def fail_parse(*args): ) assert report.verdict is Verdict.FAIL assert "private-source-path" not in report.model_dump_json() + + +@pytest.mark.skipif(not hasattr(os, "mkfifo"), reason="platform has no named pipes") +def test_source_digest_rejects_named_pipes(tmp_path): + os.mkfifo(tmp_path / "pipe") + with pytest.raises(ValueError, match="regular files"): + source_digest(tmp_path) + + +def test_source_digest_rejects_swapped_file_without_nofollow(tmp_path, monkeypatch): + package = tmp_path / "package" + package.mkdir() + source = package / "source.txt" + source.write_text("public source") + private = tmp_path / "private.txt" + private.write_text("private content") + original_open = os.open + + def swap_before_open(path, flags): + source.unlink() + source.symlink_to(private) + return original_open(path, flags) + + monkeypatch.delattr(os, "O_NOFOLLOW", raising=False) + monkeypatch.setattr(os, "open", swap_before_open) + with pytest.raises(ValueError, match="changed before"): + source_digest(package) From 74df6cc2c2cfc5f825cfdb6f4f540b5d80e07392 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Thu, 1 Oct 2026 12:12:49 +0200 Subject: [PATCH 19/29] fix: enforce HTTP collection deadlines --- src/openenv/validation/runtime/collector.py | 39 ++++------ src/openenv/validation/runtime/transport.py | 66 +++++++++++++++++ .../test_validation/test_runtime_collector.py | 72 +++++++++++++++++++ 3 files changed, 153 insertions(+), 24 deletions(-) create mode 100644 src/openenv/validation/runtime/transport.py diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index 21fc5db1ea..7260dc2bb2 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -1,7 +1,6 @@ """Bounded raw protocol collection in one OpenEnv orchestration session.""" import json -import socket import threading import time from urllib.parse import urlsplit, urlunsplit @@ -11,24 +10,12 @@ from ...core.env_server.types import WSErrorCode from .contracts import RuntimeEvidence, RuntimePlan, WireExchange +from .transport import abort_socket, http_deadline MAX_MESSAGE_BYTES = 1024 * 1024 MAX_TRACE_BYTES = 8 * 1024 * 1024 -def _abort_transport(connection): - # The send thread may hold websockets' protocol lock. Shut down the raw - # transport directly so sendall and its concurrent receiver can both exit. - try: - connection.socket.shutdown(socket.SHUT_RDWR) - except OSError: - pass # The peer or another cleanup path may already have closed it. - try: - connection.socket.close() - except OSError: - pass # A concurrent close must not replace the original operation error. - - def _bounded_call(connection, operation, timeout_s): if timeout_s <= 0: raise TimeoutError("episode deadline exceeded") @@ -37,7 +24,7 @@ def _bounded_call(connection, operation, timeout_s): def abort(): expired.set() - _abort_transport(connection) + abort_socket(connection.socket) watchdog = threading.Timer(timeout_s, abort) watchdog.daemon = True @@ -45,7 +32,7 @@ def abort(): try: operation() except KeyboardInterrupt: - _abort_transport(connection) + abort_socket(connection.socket) raise except Exception: if expired.is_set(): @@ -55,7 +42,7 @@ def abort(): watchdog.cancel() watchdog.join() if expired.is_set() or time.monotonic() >= deadline: - _abort_transport(connection) + abort_socket(connection.socket) raise TimeoutError("transport deadline exceeded") @@ -112,12 +99,16 @@ def remaining() -> float: try: with httpx.Client(trust_env=False, follow_redirects=False) as client: - with client.stream( - "GET", - base_url.rstrip("/") + "/schema", - timeout=remaining(), - headers={"Accept-Encoding": "identity"}, - ) as response: + with ( + http_deadline(remaining()) as extensions, + client.stream( + "GET", + base_url.rstrip("/") + "/schema", + timeout=remaining(), + headers={"Accept-Encoding": "identity"}, + extensions=extensions, + ) as response, + ): response.raise_for_status() # iter_bytes() transparently decompresses. Reject compressed # bodies before touching the stream so the byte budget also @@ -236,7 +227,7 @@ def exchange(operation: str, data: dict | None = None) -> dict: try: _bounded_call(connection, connection.close, 1.0) except (Exception, KeyboardInterrupt): - _abort_transport(connection) + abort_socket(connection.socket) return RuntimeEvidence( exchanges=tuple(exchanges), observation_schema_json=schema_json ) diff --git a/src/openenv/validation/runtime/transport.py b/src/openenv/validation/runtime/transport.py new file mode 100644 index 0000000000..b44c929132 --- /dev/null +++ b/src/openenv/validation/runtime/transport.py @@ -0,0 +1,66 @@ +"""Socket cancellation for bounded validation transport operations.""" + +import socket +import threading +import time +from contextlib import contextmanager + + +def abort_socket(transport_socket): + """Interrupt pending reads/writes before closing the transport.""" + try: + transport_socket.shutdown(socket.SHUT_RDWR) + except OSError: + pass # The peer or another cleanup path may already have closed it. + try: + transport_socket.close() + except OSError: + pass # Cleanup must not replace the original operation error. + + +@contextmanager +def http_deadline(timeout_s): + """Yield HTTPX trace extensions enforcing one request/body wall deadline. + + The request must establish a fresh connection. Reused clients must disable + keep-alive with `httpx.Limits(max_keepalive_connections=0)` so the trace exposes + each request's transport before headers are read. + """ + if timeout_s <= 0: + raise TimeoutError("HTTP deadline elapsed") + deadline = time.monotonic() + timeout_s + expired = threading.Event() + watched_socket = None + + def abort(): + expired.set() + if watched_socket is not None: + abort_socket(watched_socket) + + def trace(event, info): + nonlocal watched_socket + if event == "connection.connect_tcp.complete": + # Retain a duplicate descriptor: TLS wrapping detaches the original + # socket object, but shutdown on this handle still aborts the shared + # connection, including a handshake or a slow header/body read. + watched_socket = info["return_value"].get_extra_info("socket").dup() + if expired.is_set() or time.monotonic() >= deadline: + abort() + raise TimeoutError("HTTP deadline elapsed") + + watchdog = threading.Timer(timeout_s, abort) + watchdog.daemon = True + watchdog.start() + try: + yield {"trace": trace} + if expired.is_set() or time.monotonic() >= deadline: + raise TimeoutError("HTTP deadline elapsed") + except Exception: + if expired.is_set(): + raise TimeoutError("HTTP deadline elapsed") from None + raise + finally: + watchdog.cancel() + watchdog.join() + if watched_socket is not None: + watched_socket.close() diff --git a/tests/test_validation/test_runtime_collector.py b/tests/test_validation/test_runtime_collector.py index 0741e2e8a5..413b05d938 100644 --- a/tests/test_validation/test_runtime_collector.py +++ b/tests/test_validation/test_runtime_collector.py @@ -1,11 +1,16 @@ """Hostile HTTP schema responses must stay bounded before JSON/schema grading.""" import json +import socket +import threading +import time +from types import SimpleNamespace import httpx import pytest from openenv.validation.runtime import collector from openenv.validation.runtime.contracts import RuntimePlan +from openenv.validation.runtime.transport import http_deadline class TrackedStream(httpx.SyncByteStream): @@ -105,3 +110,70 @@ def test_uncompressed_schema_still_obeys_total_byte_budget(monkeypatch, plan): assert evidence.failure_phase == "schema" assert evidence.failure_reason == "schema failed (ValueError)" assert evidence.observation_schema_json is None + + +@pytest.mark.parametrize("slow_phase", ["headers", "body"]) +def test_schema_deadline_aborts_trickling_http_server(plan, slow_phase): + body = json.dumps({"observation": {"type": "object"}}).encode() + headers = ( + b"HTTP/1.1 200 OK\r\nContent-Type: application/json\r\nContent-Length: " + + str(len(body)).encode() + + b"\r\n\r\n" + ) + with socket.socket() as listener: + listener.bind(("127.0.0.1", 0)) + listener.listen() + listener.settimeout(3) + port = listener.getsockname()[1] + + def serve(): + try: + connection, _ = listener.accept() + with connection: + connection.settimeout(3) + connection.recv(65536) + if slow_phase == "body": + connection.sendall(headers) + payload = headers + body if slow_phase == "headers" else body + for byte in payload: + connection.sendall(bytes([byte])) + time.sleep(0.025) + except OSError: + pass # The deadline intentionally closes the peer transport. + + server = threading.Thread(target=serve) + server.start() + try: + started = time.monotonic() + evidence = collector.collect_runtime_evidence( + f"http://127.0.0.1:{port}", plan, episode_timeout_s=0.15 + ) + elapsed = time.monotonic() - started + finally: + server.join(timeout=4) + assert not server.is_alive() + assert evidence.failure_phase == "schema" + assert evidence.failure_reason == "schema failed (TimeoutError)" + assert elapsed < 0.6, ( + f"{slow_phase} trickle escaped the 0.15-second deadline: {elapsed:.3f}s" + ) + + +def test_http_deadline_survives_transport_socket_detach(): + peer, connection = socket.socketpair() + detached = None + try: + stream = SimpleNamespace(get_extra_info=lambda name: connection) + with pytest.raises(TimeoutError, match="HTTP deadline"): + with http_deadline(0.05) as extensions: + extensions["trace"]( + "connection.connect_tcp.complete", {"return_value": stream} + ) + # TLS wrapping similarly detaches the socket captured at connect. + detached = socket.socket(fileno=connection.detach()) + assert detached.recv(1) == b"" + finally: + peer.close() + connection.close() + if detached is not None: + detached.close() From 0cf67600d0cfeeb57b66fb36e55333e5b39aa491 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Thu, 1 Oct 2026 12:13:13 +0200 Subject: [PATCH 20/29] test: bound socket cancellation regression --- tests/test_validation/test_runtime_collector.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_validation/test_runtime_collector.py b/tests/test_validation/test_runtime_collector.py index 413b05d938..c60633f8ae 100644 --- a/tests/test_validation/test_runtime_collector.py +++ b/tests/test_validation/test_runtime_collector.py @@ -171,6 +171,7 @@ def test_http_deadline_survives_transport_socket_detach(): ) # TLS wrapping similarly detaches the socket captured at connect. detached = socket.socket(fileno=connection.detach()) + detached.settimeout(1) # Bound the regression if cancellation breaks. assert detached.recv(1) == b"" finally: peer.close() From ca2d293ef1c20ab4cbbe668e9e9002e374331c57 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Fri, 2 Oct 2026 13:29:42 +0200 Subject: [PATCH 21/29] fix: preserve startup errors --- src/openenv/validation/runtime/collector.py | 57 +++++++++++---- .../test_session_telemetry_protocol.py | 64 ++++++++++++++++- .../test_runtime_telemetry_collector.py | 70 +++++++++++++++++++ 3 files changed, 176 insertions(+), 15 deletions(-) diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index 953904bdb4..ee2ed7893e 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -6,6 +6,7 @@ from urllib.parse import urlsplit, urlunsplit import httpx +from websockets.exceptions import ConnectionClosed from websockets.sync.client import connect from ...core.env_server.types import WSErrorCode @@ -16,6 +17,16 @@ MAX_TRACE_BYTES = 8 * 1024 * 1024 +def _server_error_code(response): + if isinstance(response, dict) and response.get("type") == "error": + data = response.get("data") + code = data.get("code") if isinstance(data, dict) else None + # Only protocol constants are safe diagnostics, never subject error text. + if isinstance(code, str) and code in {item.value for item in WSErrorCode}: + return code + return None + + def _bounded_call(connection, operation, timeout_s): if timeout_s <= 0: raise TimeoutError("episode deadline exceeded") @@ -165,10 +176,30 @@ def remaining() -> float: complete = False try: + def receive_response(request): + nonlocal server_code + try: + _bounded_call( + connection, lambda: connection.send(request), remaining() + ) + except ConnectionClosed: + # Session creation can fail before the first send. recv() still + # delivers a queued error, but the failed send cannot succeed. + try: + raw = connection.recv(timeout=remaining()) + if ( + isinstance(raw, str) + and len(raw.encode()) <= MAX_TRACE_BYTES + ): + server_code = _server_error_code(json.loads(raw)) + except Exception: + pass + raise + return connection.recv(timeout=remaining()) + def telemetry_request(operation, data): request = json.dumps({"type": operation, "data": data}) - _bounded_call(connection, lambda: connection.send(request), remaining()) - raw = connection.recv(timeout=remaining()) + raw = receive_response(request) if not isinstance(raw, str) or len(raw.encode()) > MAX_TRACE_BYTES: raise ValueError("invalid telemetry response") response = json.loads(raw) @@ -188,6 +219,14 @@ def telemetry_request(operation, data): # Transport failure still aborts this same-session collection. telemetry_error = f"session telemetry failed ({type(exc).__name__})" else: + code = _server_error_code(response) + if code in { + WSErrorCode.FACTORY_ERROR, + WSErrorCode.CAPACITY_REACHED, + WSErrorCode.SESSION_ERROR, + }: + server_code = code + raise ValueError("server returned a terminal error") data = response.get("data") if ( response.get("type") == "validation_open" @@ -224,10 +263,7 @@ def exchange(operation: str, data: dict | None = None) -> dict: if data is not None: request["data"] = data request_json = json.dumps(request, allow_nan=False) - _bounded_call( - connection, lambda: connection.send(request_json), remaining() - ) - raw = connection.recv(timeout=remaining()) + raw = receive_response(request_json) if not isinstance(raw, str): raise ValueError("binary response is not the JSON protocol") if len(raw.encode("utf-8")) > MAX_MESSAGE_BYTES: @@ -259,14 +295,7 @@ def exchange(operation: str, data: dict | None = None) -> dict: ) response = json.loads(raw) if isinstance(response, dict) and response.get("type") == "error": - data = response.get("data") - code = data.get("code") if isinstance(data, dict) else None - # Only protocol constants are safe diagnostics; never echo an - # arbitrary server message or a subject-defined error code. - if isinstance(code, str) and code in { - item.value for item in WSErrorCode - }: - server_code = code + server_code = _server_error_code(response) raise ValueError("server returned an error") expected = "state" if operation == "state" else "observation" if ( diff --git a/tests/test_validation/integration/test_session_telemetry_protocol.py b/tests/test_validation/integration/test_session_telemetry_protocol.py index d4586a2665..d61b75f023 100644 --- a/tests/test_validation/integration/test_session_telemetry_protocol.py +++ b/tests/test_validation/integration/test_session_telemetry_protocol.py @@ -1,15 +1,21 @@ """Exercise the real replay WebSocket and its separate production MCP boundary.""" import json +import socket +import threading +import time import pytest +import uvicorn from fastapi import FastAPI from fastapi.testclient import TestClient from fastmcp import FastMCP -from openenv.core.env_server.http_server import HTTPEnvServer +from openenv.core.env_server.http_server import create_app, HTTPEnvServer from openenv.core.env_server.interfaces import Environment from openenv.core.env_server.types import Action, Observation, State from openenv.core.rubrics import Rubric, WeightedSum +from openenv.validation.runtime import collector +from openenv.validation.runtime.contracts import RuntimePlan TOKEN = "per-run-validation-test-token-000000000000" @@ -70,6 +76,62 @@ async def reset_async(self): return super().reset() +@pytest.mark.parametrize("closed_before_send", [False, True]) +def test_factory_error_survives_real_telemetry_handshake( + monkeypatch, closed_before_send +): + class BrokenFactory(SessionEnv): + def __init__(self): + raise RuntimeError(TOKEN) + + monkeypatch.setenv("OPENENV_VALIDATION_TOKEN", TOKEN) + app = create_app(BrokenFactory, ValueAction, Observation) + server = uvicorn.Server(uvicorn.Config(app, log_level="critical")) + with socket.socket() as listener: + listener.bind(("127.0.0.1", 0)) + port = listener.getsockname()[1] + thread = threading.Thread(target=server.run, kwargs={"sockets": [listener]}) + thread.start() + try: + deadline = time.monotonic() + 5 + while not server.started: + assert thread.is_alive() and time.monotonic() < deadline + time.sleep(0.01) + if closed_before_send: + real_connect = collector.connect + + def connect_after_close(*args, **kwargs): + connection = real_connect(*args, **kwargs) + deadline = time.monotonic() + 2 + while connection.close_code is None: + assert time.monotonic() < deadline + time.sleep(0.01) + return connection + + monkeypatch.setattr(collector, "connect", connect_after_close) + plan = RuntimePlan.model_validate( + { + "plan_schema_version": "1", + "reset": {"episode_id": "factory-failure", "seed": 1}, + "actions": [{"value": 0.5}], + } + ) + evidence = collector.collect_runtime_evidence( + f"http://127.0.0.1:{port}", + plan, + episode_timeout_s=3, + validation_token=TOKEN, + ) + assert evidence.failure_reason == "validation_open failed (FACTORY_ERROR)" + assert evidence.exchanges == () + assert evidence.telemetry_json is None + assert TOKEN not in repr(evidence) + finally: + server.should_exit = True + thread.join(timeout=5) + assert not thread.is_alive() + + def app_for(monkeypatch, env=SessionEnv, *, enabled=True, mode="simulation"): if enabled: monkeypatch.setenv("OPENENV_VALIDATION_TOKEN", TOKEN) diff --git a/tests/test_validation/test_runtime_telemetry_collector.py b/tests/test_validation/test_runtime_telemetry_collector.py index 40c6d31d8f..cf16e60c77 100644 --- a/tests/test_validation/test_runtime_telemetry_collector.py +++ b/tests/test_validation/test_runtime_telemetry_collector.py @@ -22,6 +22,7 @@ def collect( malformed=False, opening_reply=None, opening_error=None, + send_error=None, ): original = httpx.Client monkeypatch.setattr( @@ -38,6 +39,8 @@ def collect( class Connection: def send(self, raw): self.request = json.loads(raw) + if self.request["type"] == "validation_open" and send_error is not None: + raise send_error def recv(self, timeout): operation = self.request["type"] @@ -175,6 +178,73 @@ def test_opening_cancellation_is_not_downgraded_to_optional_telemetry(monkeypatc assert TOKEN not in repr(error.value.evidence) +@pytest.mark.parametrize("code", ["FACTORY_ERROR", "CAPACITY_REACHED", "SESSION_ERROR"]) +@pytest.mark.parametrize("closed_before_send", [False, True]) +def test_terminal_opening_error_preserves_only_its_safe_code( + monkeypatch, code, closed_before_send +): + result = collect( + monkeypatch, + opening_reply=json.dumps( + {"type": "error", "data": {"code": code, "message": TOKEN}} + ), + send_error=ConnectionClosedError(None, None) if closed_before_send else None, + ) + assert result.failure_reason == f"validation_open failed ({code})" + assert result.exchanges == () + assert result.telemetry_json is None + assert TOKEN not in repr(result) + + +@pytest.mark.parametrize("code", ["UNKNOWN_TYPE", "VALIDATION_ERROR", TOKEN]) +def test_optional_opening_refusal_does_not_fail_the_episode(monkeypatch, code): + result = collect( + monkeypatch, + opening_reply=json.dumps( + {"type": "error", "data": {"code": code, "message": TOKEN}} + ), + ) + assert result.failure_reason is None + assert result.telemetry_json is None + assert ( + result.telemetry_error + == "session telemetry unavailable or authorization refused" + ) + assert [row.operation for row in result.exchanges] == [ + "reset", + "state", + "step", + "state", + ] + assert TOKEN not in repr(result) + + +@pytest.mark.parametrize( + "reply", + [ + "not-json-" + TOKEN, + json.dumps( + { + "type": "validation_open", + "data": {"schema_version": 1, "capability": CAPABILITY}, + } + ), + ], +) +def test_closed_send_cannot_become_an_optional_or_successful_handshake( + monkeypatch, reply +): + result = collect( + monkeypatch, + opening_reply=reply, + send_error=ConnectionClosedError(None, None), + ) + assert result.failure_reason == "validation_open failed (ConnectionClosedError)" + assert result.exchanges == () + assert result.telemetry_json is None + assert TOKEN not in repr(result) + + @pytest.mark.parametrize("credential", [TOKEN, CAPABILITY]) @pytest.mark.parametrize("operation", ["reset", "step"]) def test_malformed_wire_cannot_retain_escaped_credentials( From b795b400968470861b51a09aec24e024672c70ab Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Fri, 2 Oct 2026 13:37:16 +0200 Subject: [PATCH 22/29] fix: align reset and unscored reward contracts --- envs/coding_tools_env/server/app.py | 6 +- envs/echo_env/server/app.py | 3 + envs/finqa_env/server/app.py | 7 +- envs/jupyter_env/server/app.py | 2 + envs/opencode_env/server/app.py | 13 +-- envs/pi_env/server/app.py | 13 +-- envs/terminus_env/server/app.py | 2 + rfcs/008-environment-auto-validation.md | 24 +++-- src/openenv/core/env_server/http_server.py | 27 ++++- src/openenv/core/env_server/types.py | 12 ++- src/openenv/core/env_server/web_interface.py | 6 +- .../validation/graders/runtime/basic.py | 17 ++- src/openenv/validation/runtime/artifacts.py | 12 +++ src/openenv/validation/runtime/collector.py | 16 ++- src/openenv/validation/runtime/contracts.py | 3 + .../validation/runtime/schema_worker.py | 19 ++-- tests/core/test_reset_observation_schema.py | 89 +++++++++++++++ .../validation/runtime/echo_canary/README.md | 14 +-- .../integration/test_echo_canary.py | 25 ++--- .../integration/test_served_probe.py | 102 ++++++++++++++++++ .../test_validation/test_runtime_artifacts.py | 49 +++++++++ .../test_validation/test_runtime_collector.py | 37 +++++++ tests/test_validation/test_runtime_grading.py | 71 +++++++++++- tests/validation_runtime/acceptance.json | 6 +- 24 files changed, 506 insertions(+), 69 deletions(-) create mode 100644 tests/core/test_reset_observation_schema.py diff --git a/envs/coding_tools_env/server/app.py b/envs/coding_tools_env/server/app.py index 51b909321c..0174a33892 100644 --- a/envs/coding_tools_env/server/app.py +++ b/envs/coding_tools_env/server/app.py @@ -13,12 +13,15 @@ from openenv.core.env_server.http_server import create_app from openenv.core.env_server.mcp_types import CallToolAction, CallToolObservation +from openenv.core.env_server.types import Observation try: from .coding_tools_env_environment import CodingToolsEnvironment from .gradio_ui import coding_tools_ui_builder except ImportError: # pragma: no cover - from server.coding_tools_env_environment import CodingToolsEnvironment # type: ignore + from server.coding_tools_env_environment import ( # type: ignore + CodingToolsEnvironment, + ) from server.gradio_ui import coding_tools_ui_builder # type: ignore @@ -44,6 +47,7 @@ def _load_env_file() -> None: CodingToolsEnvironment, CallToolAction, CallToolObservation, + reset_observation_cls=Observation, env_name="coding_tools_env", max_concurrent_envs=int(os.getenv("MAX_CONCURRENT_ENVS", "4")), gradio_builder=coding_tools_ui_builder, diff --git a/envs/echo_env/server/app.py b/envs/echo_env/server/app.py index 2cea300624..78c2a5df05 100644 --- a/envs/echo_env/server/app.py +++ b/envs/echo_env/server/app.py @@ -23,6 +23,8 @@ import os +from openenv.core.env_server.types import Observation + # Support both in-repo and standalone imports try: # In-repo imports (when running from OpenEnv repository) @@ -45,6 +47,7 @@ EchoEnvironment, CallToolAction, CallToolObservation, + reset_observation_cls=Observation, env_name="echo_env", max_concurrent_envs=max_concurrent, ) diff --git a/envs/finqa_env/server/app.py b/envs/finqa_env/server/app.py index add46afffa..21926adf13 100644 --- a/envs/finqa_env/server/app.py +++ b/envs/finqa_env/server/app.py @@ -14,6 +14,7 @@ from openenv.core.env_server.http_server import create_app from openenv.core.env_server.mcp_types import CallToolAction, CallToolObservation +from openenv.core.env_server.types import Observation from pydantic import field_validator from .finqa_environment import FinQAEnvironment @@ -44,7 +45,11 @@ def parse_arguments(cls, v: Any) -> Dict[str, Any]: app = create_app( - _env_factory, FinQACallToolAction, CallToolObservation, env_name="finqa_env" + _env_factory, + FinQACallToolAction, + CallToolObservation, + reset_observation_cls=Observation, + env_name="finqa_env", ) diff --git a/envs/jupyter_env/server/app.py b/envs/jupyter_env/server/app.py index 1fc5245267..f09373b202 100644 --- a/envs/jupyter_env/server/app.py +++ b/envs/jupyter_env/server/app.py @@ -13,6 +13,7 @@ from openenv.core.env_server.http_server import create_app from openenv.core.env_server.mcp_types import CallToolAction, CallToolObservation +from openenv.core.env_server.types import Observation try: from .gradio_ui import jupyter_ui_builder @@ -47,6 +48,7 @@ def _load_env_file() -> None: CallToolAction, CallToolObservation, env_name="jupyter_env", + reset_observation_cls=Observation, max_concurrent_envs=int(os.getenv("MAX_CONCURRENT_ENVS", "4")), gradio_builder=jupyter_ui_builder, ) diff --git a/envs/opencode_env/server/app.py b/envs/opencode_env/server/app.py index 200c7f2d77..c01b853315 100644 --- a/envs/opencode_env/server/app.py +++ b/envs/opencode_env/server/app.py @@ -29,6 +29,8 @@ import os from pathlib import Path +from openenv.core.env_server.types import Observation + def _load_env_file() -> None: """Lightweight ``.env`` loader (no python-dotenv dep). @@ -56,19 +58,13 @@ def _load_env_file() -> None: try: from openenv.core.env_server.http_server import create_app - from openenv.core.env_server.mcp_types import ( - CallToolAction, - CallToolObservation, - ) + from openenv.core.env_server.mcp_types import CallToolAction, CallToolObservation from .gradio_ui import opencode_gradio_builder from .opencode_environment import OpenCodeEnvironment except ImportError: # pragma: no cover from openenv.core.env_server.http_server import create_app - from openenv.core.env_server.mcp_types import ( - CallToolAction, - CallToolObservation, - ) + from openenv.core.env_server.mcp_types import CallToolAction, CallToolObservation from server.gradio_ui import opencode_gradio_builder # type: ignore from server.opencode_environment import OpenCodeEnvironment # type: ignore @@ -102,6 +98,7 @@ def _custom_gradio_builder( CallToolAction, CallToolObservation, env_name="opencode_env", + reset_observation_cls=Observation, max_concurrent_envs=int(os.getenv("MAX_CONCURRENT_ENVS", "4")), gradio_builder=_custom_gradio_builder, ) diff --git a/envs/pi_env/server/app.py b/envs/pi_env/server/app.py index ef22d7b1d6..a03732dd78 100644 --- a/envs/pi_env/server/app.py +++ b/envs/pi_env/server/app.py @@ -29,6 +29,8 @@ import os from pathlib import Path +from openenv.core.env_server.types import Observation + def _load_env_file() -> None: """Lightweight ``.env`` loader (no python-dotenv dep). @@ -56,19 +58,13 @@ def _load_env_file() -> None: try: from openenv.core.env_server.http_server import create_app - from openenv.core.env_server.mcp_types import ( - CallToolAction, - CallToolObservation, - ) + from openenv.core.env_server.mcp_types import CallToolAction, CallToolObservation from .gradio_ui import pi_gradio_builder from .pi_environment import PiEnvironment except ImportError: # pragma: no cover from openenv.core.env_server.http_server import create_app - from openenv.core.env_server.mcp_types import ( - CallToolAction, - CallToolObservation, - ) + from openenv.core.env_server.mcp_types import CallToolAction, CallToolObservation from server.gradio_ui import pi_gradio_builder # type: ignore from server.pi_environment import PiEnvironment # type: ignore @@ -102,6 +98,7 @@ def _custom_gradio_builder( CallToolAction, CallToolObservation, env_name="pi_env", + reset_observation_cls=Observation, max_concurrent_envs=int(os.getenv("MAX_CONCURRENT_ENVS", "4")), gradio_builder=_custom_gradio_builder, ) diff --git a/envs/terminus_env/server/app.py b/envs/terminus_env/server/app.py index 527dcbbf85..7afc3e9be9 100644 --- a/envs/terminus_env/server/app.py +++ b/envs/terminus_env/server/app.py @@ -13,6 +13,7 @@ from openenv.core.env_server.http_server import create_app from openenv.core.env_server.mcp_types import CallToolAction, CallToolObservation +from openenv.core.env_server.types import Observation try: from .gradio_ui import terminus_ui_builder @@ -45,6 +46,7 @@ def _load_env_file() -> None: CallToolAction, CallToolObservation, env_name="terminus_env", + reset_observation_cls=Observation, max_concurrent_envs=int(os.getenv("MAX_CONCURRENT_ENVS", "4")), gradio_builder=terminus_ui_builder, ) diff --git a/rfcs/008-environment-auto-validation.md b/rfcs/008-environment-auto-validation.md index 63405af7e6..43de87ee43 100644 --- a/rfcs/008-environment-auto-validation.md +++ b/rfcs/008-environment-auto-validation.md @@ -518,12 +518,24 @@ privileged oracle inputs nor host callbacks may be substituted for public action One collector owns reset, ordered actions until termination, and state reads on **one** WebSocket session. It records immutable raw request/response strings before client defaults or Pydantic coercion can hide malformed responses. Graders consume -that evidence and cannot mutate the measured episode. The advertised observation -schema is recorded alongside the transcript; reconstruct observation plus the -separate reward/done envelope before validating it. Reset reward may be null; every -step reward must be a finite JSON number, excluding booleans, within the declared -range. State must retain the requested episode identity, reset its step count to -zero and advance it coherently for successful steps. +that evidence and cannot mutate the measured episode. `/schema.observation` +describes step observations. The additive `/schema.reset_observation` field +describes reset observations; core servers publish it from an explicit optional +`reset_observation_cls`, defaulting to `observation_cls`. Environments whose reset +returns a distinct model must declare it. For legacy servers that omit the field, +the validator applies the step schema to reset as well. An explicitly malformed +reset schema is a failure, never a reason to fall back. Both schemas are recorded +alongside the transcript; reconstruct observation plus the separate reward/done +envelope before validating against the operation's schema. Neither schema may +retrieve non-local references. + +Reset reward may be null. A step with `done: false` may have null reward to denote +that no score was emitted. A terminal step must emit a finite JSON number within +the declared range. Every numeric reset or step reward is checked against that +range; booleans are invalid under this Level Two profile. This does not narrow the +core Observation model's backward-compatible reward type. State must retain the +requested episode identity, reset its step count to zero and advance it coherently +for successful steps. ### Startup, policy and provider supervision diff --git a/src/openenv/core/env_server/http_server.py b/src/openenv/core/env_server/http_server.py index ecc2c5c2d0..92b40d642a 100644 --- a/src/openenv/core/env_server/http_server.py +++ b/src/openenv/core/env_server/http_server.py @@ -179,6 +179,8 @@ def __init__( concurrency_config: Optional[ConcurrencyConfig] = None, env_name: Optional[str] = None, state_cls: Type[State] = State, + *, + reset_observation_cls: Optional[Type[Observation]] = None, ): """ Initialize HTTP server wrapper. @@ -190,7 +192,10 @@ def __init__( action_cls (`Type[Action]`): The `Action` subclass this environment expects. observation_cls (`Type[Observation]`): - The `Observation` subclass this environment returns. + The `Observation` subclass returned by step. + reset_observation_cls (`Type[Observation]`, *optional*): + The reset observation model published in `/schema`. Defaults to + `observation_cls`; declare a distinct model when reset differs. max_concurrent_envs (`int`, *optional*): Maximum number of concurrent WebSocket sessions. Mutually exclusive with `concurrency_config`. @@ -246,6 +251,7 @@ def __init__( self.action_cls = action_cls self.observation_cls = observation_cls + self.reset_observation_cls = reset_observation_cls or observation_cls self.state_cls = state_cls self.env_name = env_name or self._default_env_name() @@ -1659,7 +1665,8 @@ def get_metadata_handler() -> EnvironmentMetadata: Returns a combined schema object containing: - **action**: JSON schema for actions accepted by this environment -- **observation**: JSON schema for observations returned by this environment +- **observation**: JSON schema for observations returned by step +- **reset_observation**: JSON schema for observations returned by reset - **state**: JSON schema for environment state objects This is more efficient than calling individual schema endpoints and provides @@ -1694,6 +1701,7 @@ async def get_schemas() -> SchemaResponse: return SchemaResponse( action=self.action_cls.model_json_schema(), observation=self.observation_cls.model_json_schema(), + reset_observation=self.reset_observation_cls.model_json_schema(), state=self.state_cls.model_json_schema(), ) @@ -1994,6 +2002,7 @@ def create_app( state_cls: Type[State] = State, *, mode: Optional[ServerMode | str] = None, + reset_observation_cls: Optional[Type[Observation]] = None, ) -> FastAPI: """ Create a FastAPI application with or without web interface. @@ -2007,7 +2016,10 @@ def create_app( action_cls (`Type[Action]`): The Action subclass this environment expects. observation_cls (`Type[Observation]`): - The Observation subclass this environment returns. + The Observation subclass returned by step. + reset_observation_cls (`Type[Observation]`, *optional*): + The reset observation model published in `/schema`. Defaults to + `observation_cls`; declare a distinct model when reset differs. env_name (`str`, *optional*): Environment name for README loading. max_concurrent_envs (`int`, *optional*): @@ -2062,6 +2074,7 @@ def create_app( max_concurrent_envs, concurrency_config, state_cls=state_cls, + reset_observation_cls=reset_observation_cls, gradio_builder=gradio_builder, custom_tab_name=custom_tab_name, custom_tab_primary=custom_tab_primary, @@ -2079,6 +2092,7 @@ def create_app( concurrency_config, env_name=env_name, state_cls=state_cls, + reset_observation_cls=reset_observation_cls, mode=mode, ) @@ -2093,6 +2107,7 @@ def create_fastapi_app( state_cls: Type[State] = State, *, mode: Optional[ServerMode | str] = None, + reset_observation_cls: Optional[Type[Observation]] = None, ) -> FastAPI: """ Create a FastAPI application with comprehensive documentation. @@ -2103,7 +2118,10 @@ def create_fastapi_app( action_cls (`Type[Action]`): The Action subclass this environment expects. observation_cls (`Type[Observation]`): - The Observation subclass this environment returns. + The Observation subclass returned by step. + reset_observation_cls (`Type[Observation]`, *optional*): + The reset observation model published in `/schema`. Defaults to + `observation_cls`; declare a distinct model when reset differs. max_concurrent_envs (`int`, *optional*): Maximum concurrent WebSocket sessions. Mutually exclusive with `concurrency_config`. @@ -2197,6 +2215,7 @@ def create_fastapi_app( concurrency_config=concurrency_config, env_name=env_name, state_cls=state_cls, + reset_observation_cls=reset_observation_cls, ) if mode is None: mode = os.environ.get("OPENENV_MODE", ServerMode.SIMULATION.value) diff --git a/src/openenv/core/env_server/types.py b/src/openenv/core/env_server/types.py index 14fbdcb986..3ccc69bcb5 100644 --- a/src/openenv/core/env_server/types.py +++ b/src/openenv/core/env_server/types.py @@ -3,7 +3,7 @@ from enum import Enum from typing import Annotated, Any, Dict, Literal, Optional, Union -from pydantic import BaseModel, ConfigDict, Field, model_validator +from pydantic import BaseModel, ConfigDict, Field, model_serializer, model_validator # Type aliases @@ -236,6 +236,16 @@ class SchemaResponse(BaseMessage): state: Dict[str, Any] = Field( description="JSON schema for environment state objects" ) + reset_observation: Optional[Dict[str, Any]] = Field( + default=None, description="JSON schema for observations returned by reset" + ) + + @model_serializer(mode="wrap") + def _omit_undeclared_reset_schema(self, handler): + data = handler(self) + if self.reset_observation is None: + data.pop("reset_observation", None) + return data class HealthResponse(BaseMessage): diff --git a/src/openenv/core/env_server/web_interface.py b/src/openenv/core/env_server/web_interface.py index 1c0ef4e5f9..772e48c273 100644 --- a/src/openenv/core/env_server/web_interface.py +++ b/src/openenv/core/env_server/web_interface.py @@ -437,6 +437,7 @@ def create_web_interface_app( state_cls: Type[State] = State, *, mode: Optional[Any] = None, + reset_observation_cls: Optional[Type[Observation]] = None, ) -> FastAPI: """ Create a FastAPI application with web interface for the given environment. @@ -444,7 +445,9 @@ def create_web_interface_app( Args: env: The Environment instance to serve action_cls: The Action subclass this environment expects - observation_cls: The Observation subclass this environment returns + observation_cls: The Observation subclass returned by step + reset_observation_cls: The reset observation model published in /schema. + Defaults to observation_cls. env_name: Optional environment name for README loading max_concurrent_envs: Maximum concurrent WebSocket sessions concurrency_config: Optional ConcurrencyConfig for advanced concurrency settings @@ -487,6 +490,7 @@ def create_web_interface_app( concurrency_config, env_name=env_name, state_cls=state_cls, + reset_observation_cls=reset_observation_cls, mode=mode, ) diff --git a/src/openenv/validation/graders/runtime/basic.py b/src/openenv/validation/graders/runtime/basic.py index 540b362f51..79a95a33ac 100644 --- a/src/openenv/validation/graders/runtime/basic.py +++ b/src/openenv/validation/graders/runtime/basic.py @@ -77,7 +77,7 @@ def run(self, subject) -> CheckResult: class RewardWellFormedGrader(_RuntimeGrader): - """Require finite numeric step rewards within the manifest's declared range.""" + """Allow unscored nonterminal steps; bound every emitted numeric reward.""" check_id = "runtime.reward_well_formed" @@ -97,7 +97,9 @@ def check(self, subject, evidence) -> list[str]: problems.append(f"exchange {index}: missing reward") continue reward = data["reward"] - if reward is None and exchange.operation == "reset": + if reward is None and ( + exchange.operation == "reset" or data.get("done") is False + ): continue if ( type(reward) not in (int, float) @@ -122,6 +124,10 @@ def check(self, subject, evidence) -> list[str]: # Submitted schemas must not cause host-side HTTP/file retrieval. if _remote_reference(schema): return ["observation schema has a non-local reference"] + if evidence.reset_observation_schema_json is not None: + reset_schema = json.loads(evidence.reset_observation_schema_json) + if _remote_reference(reset_schema): + return ["reset observation schema has a non-local reference"] problems = [] observations = [] count = 0 @@ -144,7 +150,11 @@ def check(self, subject, evidence) -> list[str]: # boundary: normalizing numbers such as 1e9 can inflate a valid # episode beyond the worker's input limit. observations.append( - {"index": index, "response_json": exchange.response_json} + { + "index": index, + "operation": exchange.operation, + "response_json": exchange.response_json, + } ) if count == 0: problems.append("no observations were measured") @@ -157,6 +167,7 @@ def check(self, subject, evidence) -> list[str]: input=json.dumps( { "schema_json": evidence.observation_schema_json, + "reset_schema_json": evidence.reset_observation_schema_json, "observations": observations, }, ensure_ascii=False, diff --git a/src/openenv/validation/runtime/artifacts.py b/src/openenv/validation/runtime/artifacts.py index 1e2079c37d..e69f9fcfd2 100644 --- a/src/openenv/validation/runtime/artifacts.py +++ b/src/openenv/validation/runtime/artifacts.py @@ -91,12 +91,23 @@ def write_runtime_bundle( schema = json.loads(evidence.observation_schema_json) except (ValueError, RecursionError): schema_parse_failed = True + reset_schema = None + reset_schema_parse_failed = False + if evidence.reset_observation_schema_json is not None: + try: + reset_schema = json.loads(evidence.reset_observation_schema_json) + except (ValueError, RecursionError): + reset_schema_parse_failed = True collector_metadata = { "evidence_schema_version": "1", "trace_file": "collector-trace.json", "observation_schema": schema, "schema_available": evidence.observation_schema_json is not None, "schema_parse_failed": schema_parse_failed, + "reset_observation_schema": reset_schema, + "reset_schema_available": evidence.reset_observation_schema_json + is not None, + "reset_schema_parse_failed": reset_schema_parse_failed, "omitted_trace_fields": omitted_fields, "failure_phase": evidence.failure_phase, "failure_reason": evidence.failure_reason, @@ -106,6 +117,7 @@ def write_runtime_bundle( collector_metadata["redacted"] = ( bool(omitted_fields) or schema_parse_failed + or reset_schema_parse_failed or _redact(trace) != trace or _redact(collector_metadata) != collector_metadata ) diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index 7260dc2bb2..4fd853094f 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -84,7 +84,7 @@ def collect_runtime_evidence( """ deadline = time.monotonic() + episode_timeout_s exchanges = [] - schema_json = None + schema_json = reset_schema_json = None phase = "schema" trace_bytes = 0 server_code = None @@ -134,6 +134,14 @@ def remaining() -> float: separators=(",", ":"), ) + if "reset_observation" in schema: + reset_schema_json = json.dumps( + schema["reset_observation"], + allow_nan=False, + ensure_ascii=False, + separators=(",", ":"), + ) + endpoint = urlsplit(base_url) ws_url = urlunsplit( ( @@ -229,13 +237,16 @@ def exchange(operation: str, data: dict | None = None) -> dict: except (Exception, KeyboardInterrupt): abort_socket(connection.socket) return RuntimeEvidence( - exchanges=tuple(exchanges), observation_schema_json=schema_json + exchanges=tuple(exchanges), + observation_schema_json=schema_json, + reset_observation_schema_json=reset_schema_json, ) except KeyboardInterrupt: raise RuntimeCollectionInterrupted( RuntimeEvidence( exchanges=tuple(exchanges), observation_schema_json=schema_json, + reset_observation_schema_json=reset_schema_json, failure_phase=phase, failure_reason=f"{phase} failed (KeyboardInterrupt)", ) @@ -245,6 +256,7 @@ def exchange(operation: str, data: dict | None = None) -> dict: return RuntimeEvidence( exchanges=tuple(exchanges), observation_schema_json=schema_json, + reset_observation_schema_json=reset_schema_json, failure_phase=phase, failure_reason=f"{phase} failed ({server_code or type(exc).__name__})", ) diff --git a/src/openenv/validation/runtime/contracts.py b/src/openenv/validation/runtime/contracts.py index 3d84e7be68..9440783f96 100644 --- a/src/openenv/validation/runtime/contracts.py +++ b/src/openenv/validation/runtime/contracts.py @@ -265,9 +265,12 @@ class RuntimeEvidence: Collector phase that failed; a truncated transcript cannot pass silently. failure_reason (`str`, *optional*): Credential-safe explanation of the collection failure. + reset_observation_schema_json (`str`, *optional*): + Explicit `/schema` reset_observation value; absent means use the step schema. """ exchanges: tuple[WireExchange, ...] = () observation_schema_json: str | None = None failure_phase: str | None = None failure_reason: str | None = None + reset_observation_schema_json: str | None = None diff --git a/src/openenv/validation/runtime/schema_worker.py b/src/openenv/validation/runtime/schema_worker.py index b5ff8ef1de..62e10978f7 100644 --- a/src/openenv/validation/runtime/schema_worker.py +++ b/src/openenv/validation/runtime/schema_worker.py @@ -7,7 +7,7 @@ from referencing import Registry # Quoting the 8 MiB raw trace costs at most 16 MiB. The remainder covers the -# 1 MiB schema after conservative 6x JSON normalization and 2x quoting, plus row +# 1 MiB combined schemas after conservative 6x normalization and 2x quoting, plus row # metadata. Limits apply to bytes, independently of the text stream's encoding. MAX_INPUT_BYTES = 32 * 1024 * 1024 @@ -29,16 +29,21 @@ def main(): payload = json.loads(raw) problems = [] try: - schema = json.loads(payload["schema_json"]) - Draft202012Validator.check_schema(schema) - # An empty registry has no retrieval callback: unresolved references - # cannot trigger host filesystem or network access. - validator = Draft202012Validator(schema, registry=Registry()) + schemas = {"step": payload["schema_json"]} + if payload.get("reset_schema_json") is not None: + schemas["reset"] = payload["reset_schema_json"] + validators = {} + for operation, raw_schema in schemas.items(): + schema = json.loads(raw_schema) + Draft202012Validator.check_schema(schema) + # An empty registry prevents filesystem and network retrieval. + validators[operation] = Draft202012Validator(schema, registry=Registry()) + validators.setdefault("reset", validators["step"]) for row in payload["observations"]: data = json.loads(row["response_json"])["data"] observation = dict(data["observation"]) observation.update(reward=data["reward"], done=data["done"]) - for error in validator.iter_errors(observation): + for error in validators[row["operation"]].iter_errors(observation): location = "/".join(str(x) for x in error.absolute_schema_path)[:160] missing = "" if error.validator == "required": diff --git a/tests/core/test_reset_observation_schema.py b/tests/core/test_reset_observation_schema.py new file mode 100644 index 0000000000..9f2ff6e7bc --- /dev/null +++ b/tests/core/test_reset_observation_schema.py @@ -0,0 +1,89 @@ +"""Reset schemas are explicit declarations, independent of step observations.""" + +import json + +import pytest +from fastapi import FastAPI +from fastapi.testclient import TestClient +from openenv.core.env_server.http_server import create_app, create_fastapi_app +from openenv.core.env_server.interfaces import Environment +from openenv.core.env_server.types import Action, Observation, SchemaResponse, State + + +class StepObservation(Observation): + value: int + + +class ResetObservation(Observation): + ready: bool + + +class DistinctResetEnvironment(Environment[Action, StepObservation, State]): + def reset(self, seed=None, episode_id=None): + return ResetObservation(ready=True, reward=None) + + def step(self, action): + return StepObservation(value=1, reward=1.0, done=True) + + @property + def state(self): + return State(episode_id="reset-schema", step_count=0) + + +@pytest.mark.parametrize("web_enabled", [False, True]) +def test_explicit_reset_model_is_published_and_matches_raw_reset( + monkeypatch, web_enabled +): + monkeypatch.setenv("ENABLE_WEB_INTERFACE", str(web_enabled).lower()) + app = create_app( + DistinctResetEnvironment, + Action, + StepObservation, + reset_observation_cls=ResetObservation, + ) + with TestClient(app) as client: + schema = client.get("/schema").json() + assert schema["reset_observation"] == ResetObservation.model_json_schema() + assert schema["observation"] == StepObservation.model_json_schema() + with client.websocket_connect("/ws") as ws: + ws.send_json({"type": "reset", "data": {}}) + response = ws.receive_json() + assert response["data"]["observation"]["ready"] is True + assert "value" not in response["data"]["observation"] + + +def test_default_reset_schema_keeps_declared_model_without_creating_environment(): + def factory(): + raise AssertionError("Schema discovery must not instantiate the environment") + + with TestClient(create_fastapi_app(factory, Action, StepObservation)) as client: + schema = client.get("/schema").json() + assert schema["reset_observation"] == schema["observation"] + assert schema["reset_observation"]["required"] == ["value"] + + +def test_schema_response_still_accepts_legacy_three_schema_payload(): + response = SchemaResponse(action={}, observation={}, state={}) + assert response.reset_observation is None + assert "reset_observation" not in response.model_dump() + assert "reset_observation" not in json.loads(response.model_dump_json()) + app = FastAPI() + + @app.get("/schema", response_model=SchemaResponse) + def schema(): + return response + + with TestClient(app) as client: + assert "reset_observation" not in client.get("/schema").json() + + +def test_reference_echo_explicitly_advertises_its_base_reset_observation(): + from echo_env.server.app import app + + with TestClient(app) as client: + schema = client.get("/schema").json() + assert schema["reset_observation"] == Observation.model_json_schema() + assert "tool_name" in schema["observation"]["required"] + with client.websocket_connect("/ws") as ws: + ws.send_json({"type": "reset", "data": {"episode_id": "reset-schema"}}) + assert "tool_name" not in ws.receive_json()["data"]["observation"] diff --git a/tests/fixtures/validation/runtime/echo_canary/README.md b/tests/fixtures/validation/runtime/echo_canary/README.md index cf243fd961..3db95588ca 100644 --- a/tests/fixtures/validation/runtime/echo_canary/README.md +++ b/tests/fixtures/validation/runtime/echo_canary/README.md @@ -9,10 +9,10 @@ hash-checked wheelhouse and offline installation. Its image recipe changes only the subject COPY and entrypoint. The validator never imports Echo on the host. The acceptance test requires real tool responses and a continuing episode with -state counts 0, 1 and 2. The current Echo implementation emits null step rewards -and resets with a base observation despite advertising CallToolObservation as its -schema. Consequently the expected validator result is FAIL with explicit reward -and schema findings, while startup and state checks pass. A passing acceptance -test proves those compatibility findings are observed; it does not certify Echo -or complete Level 2 validation. When the environment contract changes, update the -expected findings together with evidence of the intended behavior. +state counts 0, 1 and 2. Echo explicitly advertises its base reset observation +separately from CallToolObservation step responses. Its null rewards denote +unscored, nonterminal tool calls, permitted by the Level Two reward profile. +Startup, state, reward and observation schema checks must pass. Unimplemented +checks still produce SKIP and the report remains WARN; this does not certify Echo +or complete Level 2 validation. Terminal null rewards, malformed reset observations +and legacy single-schema mismatches remain failures in dedicated regressions. diff --git a/tests/test_validation/integration/test_echo_canary.py b/tests/test_validation/integration/test_echo_canary.py index c3c3859f6a..054751cf2c 100644 --- a/tests/test_validation/integration/test_echo_canary.py +++ b/tests/test_validation/integration/test_echo_canary.py @@ -12,7 +12,7 @@ pytestmark = pytest.mark.docker -def test_reference_echo_replays_real_tools_and_exposes_contract_findings(tmp_path): +def test_reference_echo_validates_reset_and_unscored_steps(tmp_path): configured = os.environ.get("OPENENV_VALIDATION_ECHO_CONTEXT") if not configured: if os.environ.get("OPENENV_REQUIRE_DOCKER") == "1": @@ -34,8 +34,8 @@ def test_reference_echo_replays_real_tools_and_exposes_contract_findings(tmp_pat expected = { "runtime.startup": "pass", "runtime.state_contract": "pass", - "runtime.reward_well_formed": "fail", - "runtime.observation_schema": "fail", + "runtime.reward_well_formed": "pass", + "runtime.observation_schema": "pass", } (evidence_root / "cli/echo_canary/compatibility-findings.json").write_text( json.dumps( @@ -51,7 +51,7 @@ def test_reference_echo_replays_real_tools_and_exposes_contract_findings(tmp_pat } for check_id in expected }, - "interpretation": "Expected contract findings; Echo is not certified.", + "interpretation": "Implemented runtime checks pass; skipped checks keep Echo uncertified.", }, indent=2, sort_keys=True, @@ -94,19 +94,14 @@ def test_reference_echo_replays_real_tools_and_exposes_contract_findings(tmp_pat assert step["reward"] is None assert step["done"] is False - # Preserve genuine compatibility findings instead of modifying Echo or - # accepting a permissive validator result merely to turn this canary green. - assert result.returncode == 1 - assert report["verdict"] == "fail" - assert checks["runtime.reward_well_formed"]["status"] == "fail" - assert checks["runtime.observation_schema"]["status"] == "fail" - assert any( - "finite and in range" in entry - for entry in checks["runtime.reward_well_formed"]["evidence"] - ) - assert checks["runtime.observation_schema"]["evidence"] + assert result.returncode == 0 + assert report["verdict"] == "warn" + assert checks["runtime.reward_well_formed"]["status"] == "pass" + assert checks["runtime.observation_schema"]["status"] == "pass" assert "tool_name" not in trace[0]["response_json"]["data"]["observation"] metadata = artifacts["collector-evidence.json"] assert metadata["complete"] is True assert metadata["failure_phase"] is None assert "tool_name" in metadata["observation_schema"]["required"] + assert metadata["reset_schema_available"] is True + assert "tool_name" not in metadata["reset_observation_schema"].get("required", []) diff --git a/tests/test_validation/integration/test_served_probe.py b/tests/test_validation/integration/test_served_probe.py index a02f8d1594..57eb591041 100644 --- a/tests/test_validation/integration/test_served_probe.py +++ b/tests/test_validation/integration/test_served_probe.py @@ -1,9 +1,15 @@ """Test fixture faults over the production OpenEnv WebSocket endpoint.""" import importlib.util +import json +import socket +import threading +import time from pathlib import Path +from types import SimpleNamespace import pytest +import uvicorn from starlette.testclient import TestClient @@ -64,3 +70,99 @@ def test_fault_is_visible_in_raw_wire_response(mode): def test_failed_start_is_explicit(): with pytest.raises(RuntimeError, match="Deliberate startup failure"): fixture.make_app("startup_failure") + + +@pytest.mark.parametrize("legacy_schema", [False, True]) +def test_real_echo_reset_schema_and_unscored_rewards(legacy_schema): + from echo_env.server.echo_environment import EchoEnvironment + from openenv.core.env_server.http_server import create_fastapi_app + from openenv.core.env_server.mcp_types import CallToolAction, CallToolObservation + from openenv.core.env_server.types import Observation + from openenv.validation.graders.runtime import ( + ObservationSchemaGrader, + RewardWellFormedGrader, + StateContractGrader, + ) + from openenv.validation.runtime.collector import collect_runtime_evidence + from openenv.validation.runtime.contracts import RuntimePlan + + app = create_fastapi_app( + EchoEnvironment, + CallToolAction, + CallToolObservation, + reset_observation_cls=Observation, + ) + + async def served_app(scope, receive, send): + async def legacy_send(message): + # Reproduce an older server's actual single-schema HTTP response. + if message["type"] == "http.response.start": + message = { + **message, + "headers": [ + (key, value) + for key, value in message["headers"] + if key != b"content-length" + ], + } + elif message["type"] == "http.response.body": + schema = json.loads(message["body"]) + schema.pop("reset_observation") + message = {**message, "body": json.dumps(schema).encode()} + await send(message) + + await app( + scope, + receive, + legacy_send if legacy_schema and scope.get("path") == "/schema" else send, + ) + + plan = RuntimePlan.model_validate( + { + "plan_schema_version": "1", + "reset": {"episode_id": "echo-contract", "seed": 42}, + "actions": [ + { + "tool_name": "echo_message", + "arguments": {"message": "real contract probe"}, + } + ], + } + ) + with socket.socket() as listener: + listener.bind(("127.0.0.1", 0)) + port = listener.getsockname()[1] + server = uvicorn.Server( + uvicorn.Config(served_app, log_level="error", lifespan="on") + ) + thread = threading.Thread(target=server.run, kwargs={"sockets": [listener]}) + thread.start() + try: + deadline = time.monotonic() + 5 + while ( + not server.started and thread.is_alive() and time.monotonic() < deadline + ): + time.sleep(0.01) + assert server.started + evidence = collect_runtime_evidence( + f"http://127.0.0.1:{port}", plan, episode_timeout_s=5 + ) + finally: + server.should_exit = True + thread.join(timeout=5) + assert not thread.is_alive() + assert evidence.failure_reason is None + assert (evidence.reset_observation_schema_json is None) is legacy_schema + step = next(row for row in evidence.exchanges if row.operation == "step") + response = json.loads(step.response_json)["data"] + assert response["reward"] is None and response["done"] is False + assert "real contract probe" in json.dumps(response["observation"]["result"]) + subject = SimpleNamespace( + runtime_evidence=evidence, + manifest=SimpleNamespace(reward=SimpleNamespace(range=(0, 1))), + ) + assert RewardWellFormedGrader().run(subject).status.value == "pass" + assert StateContractGrader().run(subject).status.value == "pass" + assert ObservationSchemaGrader().run(subject).status.value == ( + "fail" if legacy_schema else "pass" + ) diff --git a/tests/test_validation/test_runtime_artifacts.py b/tests/test_validation/test_runtime_artifacts.py index fcb2cc8c36..d54af80eba 100644 --- a/tests/test_validation/test_runtime_artifacts.py +++ b/tests/test_validation/test_runtime_artifacts.py @@ -2,6 +2,7 @@ import json from dataclasses import replace +import pytest from conftest import load_fixture_manifest from openenv.validation.graders import Subject from openenv.validation.graders.runtime import ( @@ -87,6 +88,11 @@ def rebuild(directory): if metadata["schema_available"] else None ), + reset_observation_schema_json=( + json.dumps(metadata["reset_observation_schema"]) + if metadata["reset_schema_available"] + else None + ), failure_phase=metadata["failure_phase"], failure_reason=metadata["failure_reason"], ) @@ -242,3 +248,46 @@ def test_redaction_preserves_token_metadata_and_filters_secret_keys(): assert filtered["prompt_token_ids"] == [1, 2, 3] assert filtered["access_token"] == "[REDACTED]" assert secret not in json.dumps(filtered) + + +@pytest.mark.parametrize( + "reset_schema", [None, "null", '{"required":["missing-reset-field"]}'] +) +def test_reset_schema_bundle_preserves_selection_and_grader_result( + tmp_path, reset_schema +): + original = replace(measured(), reset_observation_schema_json=reset_schema) + validation_report = report() + write_runtime_bundle(tmp_path, validation_report, evidence=original) + replay = rebuild(tmp_path) + if reset_schema is None: + assert replay.reset_observation_schema_json is None + else: + assert json.loads(replay.reset_observation_schema_json) == json.loads( + reset_schema + ) + subject = Subject( + tmp_path, validation_report.manifest, None, None, tmp_path, original + ) + expected = ObservationSchemaGrader().run(subject) + actual = ObservationSchemaGrader().run(replace(subject, runtime_evidence=replay)) + assert actual.status == expected.status + assert actual.evidence == expected.evidence + + +@pytest.mark.parametrize( + "schema,parse_failed", + [("not JSON", True), ('{"description":"hf_abcdefghijk123456789"}', False)], +) +def test_reset_schema_omission_or_redaction_is_visible(tmp_path, schema, parse_failed): + write_runtime_bundle( + tmp_path, + report(), + evidence=replace(measured(), reset_observation_schema_json=schema), + ) + text = (tmp_path / "collector-evidence.json").read_text() + metadata = json.loads(text) + assert metadata["reset_schema_available"] is True + assert metadata["reset_schema_parse_failed"] is parse_failed + assert metadata["redacted"] is True + assert "hf_abcdefghijk123456789" not in text diff --git a/tests/test_validation/test_runtime_collector.py b/tests/test_validation/test_runtime_collector.py index c60633f8ae..6c9ee8f2f1 100644 --- a/tests/test_validation/test_runtime_collector.py +++ b/tests/test_validation/test_runtime_collector.py @@ -178,3 +178,40 @@ def test_http_deadline_survives_transport_socket_detach(): connection.close() if detached is not None: detached.close() + + +@pytest.mark.parametrize( + "schema", + [{}, {"reset_observation": None}, {"reset_observation": {"required": ["ready"]}}], +) +def test_reset_schema_presence_survives_later_collection_failure( + monkeypatch, plan, schema +): + stream = TrackedStream( + json.dumps({"observation": {"type": "object"}, **schema}).encode() + ) + schema_transport(monkeypatch, stream) + evidence = collector.collect_runtime_evidence( + "http://127.0.0.1:8000", plan, episode_timeout_s=2 + ) + assert evidence.failure_phase == "connect" + assert evidence.reset_observation_schema_json == ( + json.dumps(schema["reset_observation"], separators=(",", ":")) + if "reset_observation" in schema + else None + ) + + +def test_reset_schema_shares_entire_response_byte_budget(monkeypatch, plan): + stream = TrackedStream( + json.dumps( + {"observation": {}, "reset_observation": {"description": "x" * 100}} + ).encode() + ) + schema_transport(monkeypatch, stream) + monkeypatch.setattr(collector, "MAX_MESSAGE_BYTES", 64) + evidence = collector.collect_runtime_evidence( + "http://127.0.0.1:8000", plan, episode_timeout_s=2 + ) + assert evidence.failure_phase == "schema" + assert evidence.reset_observation_schema_json is None diff --git a/tests/test_validation/test_runtime_grading.py b/tests/test_validation/test_runtime_grading.py index b89e6f1840..3c8dba8591 100644 --- a/tests/test_validation/test_runtime_grading.py +++ b/tests/test_validation/test_runtime_grading.py @@ -99,7 +99,6 @@ def test_good_measured_session_passes_each_basic_contract(tmp_path, grader): [ True, False, - None, "0.5", [], {}, @@ -111,15 +110,81 @@ def test_good_measured_session_passes_each_basic_contract(tmp_path, grader): 1.1, ], ) -def test_step_rewards_are_checked_without_coercion(tmp_path, reward): +@pytest.mark.parametrize("done", [False, True]) +def test_step_rewards_are_checked_without_coercion(tmp_path, reward, done): rows = mutate_response( - good_rows(), 2, lambda response: response["data"].update(reward=reward) + good_rows(), + 2, + lambda response: response["data"].update(reward=reward, done=done), ) result = RewardWellFormedGrader().run(subject_with(tmp_path, rows)) assert result.status is CheckStatus.FAIL assert any("reward" in message for message in result.evidence) +@pytest.mark.parametrize("done", [False, True, None, 0]) +def test_null_step_reward_requires_explicit_nonterminal_observation(tmp_path, done): + rows = mutate_response( + good_rows(), 2, lambda response: response["data"].update(reward=None, done=done) + ) + result = RewardWellFormedGrader().run(subject_with(tmp_path, rows)) + assert result.status is (CheckStatus.PASS if done is False else CheckStatus.FAIL) + + +def test_explicit_reset_schema_is_independent_of_step_schema(tmp_path): + rows = mutate_response( + good_rows(), + 0, + lambda response: response["data"].update(observation={"ready": True}), + ) + subject = subject_with(tmp_path, rows) + # A legacy single-schema server remains strict about reset fields. + assert ObservationSchemaGrader().run(subject).status is CheckStatus.FAIL + subject = replace( + subject, + runtime_evidence=replace( + subject.runtime_evidence, + reset_observation_schema_json=json.dumps( + { + "type": "object", + "properties": {"ready": {"type": "boolean"}}, + "required": ["ready"], + } + ), + ), + ) + assert ObservationSchemaGrader().run(subject).status is CheckStatus.PASS + malformed = mutate_response( + rows, 0, lambda response: response["data"].update(observation={"ready": "yes"}) + ) + subject = replace( + subject, + runtime_evidence=replace(subject.runtime_evidence, exchanges=tuple(malformed)), + ) + assert ObservationSchemaGrader().run(subject).status is CheckStatus.FAIL + + +@pytest.mark.parametrize( + "schema", + [ + None, + [], + {"type": "invalid"}, + False, + {"$ref": "https://example.invalid/reset.json"}, + ], +) +def test_explicit_invalid_or_rejecting_reset_schema_cannot_fall_back(tmp_path, schema): + subject = subject_with(tmp_path) + subject = replace( + subject, + runtime_evidence=replace( + subject.runtime_evidence, reset_observation_schema_json=json.dumps(schema) + ), + ) + assert ObservationSchemaGrader().run(subject).status is CheckStatus.FAIL + + def test_missing_step_reward_is_an_explicit_failure(tmp_path): rows = mutate_response( good_rows(), 2, lambda response: response["data"].pop("reward") diff --git a/tests/validation_runtime/acceptance.json b/tests/validation_runtime/acceptance.json index 6ab4830800..f68cc85901 100644 --- a/tests/validation_runtime/acceptance.json +++ b/tests/validation_runtime/acceptance.json @@ -7,13 +7,15 @@ "tests.test_validation.integration.test_served_probe::test_fault_is_visible_in_raw_wire_response[bad_observation]", "tests.test_validation.integration.test_served_probe::test_fault_is_visible_in_raw_wire_response[missing_done]", "tests.test_validation.integration.test_served_probe::test_fault_is_visible_in_raw_wire_response[bad_state]", - "tests.test_validation.integration.test_served_probe::test_failed_start_is_explicit" + "tests.test_validation.integration.test_served_probe::test_failed_start_is_explicit", + "tests.test_validation.integration.test_served_probe::test_real_echo_reset_schema_and_unscored_rewards[False]", + "tests.test_validation.integration.test_served_probe::test_real_echo_reset_schema_and_unscored_rewards[True]" ], "docker": [ "tests.test_validation.integration.test_docker_lifecycle::test_docker_lifecycle_effective_limits_and_owned_cleanup", "tests.test_validation.integration.test_docker_lifecycle::test_docker_unhealthy_startup_removes_container", "tests.test_validation.integration.test_docker_lifecycle::test_docker_exec_timeout_removes_process_tree", - "tests.test_validation.integration.test_echo_canary::test_reference_echo_replays_real_tools_and_exposes_contract_findings", + "tests.test_validation.integration.test_echo_canary::test_reference_echo_validates_reset_and_unscored_steps", "tests.test_validation.integration.test_runtime_cli::test_cli_runtime_contract_findings[good-None]", "tests.test_validation.integration.test_runtime_cli::test_cli_runtime_contract_findings[slow_step-None]", "tests.test_validation.integration.test_runtime_cli::test_cli_runtime_contract_findings[bad_reward-runtime.reward_well_formed]", From c7c41013bbe76620401e791bb31161c7b675b1f5 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Fri, 2 Oct 2026 13:38:59 +0200 Subject: [PATCH 23/29] fix: protect reset schema evidence --- src/openenv/validation/runtime/collector.py | 5 ++++- tests/test_validation/test_runtime_collector.py | 15 ++++++++++++--- tests/validation_runtime/acceptance.json | 2 ++ 3 files changed, 18 insertions(+), 4 deletions(-) diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index 7d50dc7950..dbc42c7e36 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -154,12 +154,15 @@ def remaining() -> float: schema_json = schema_payload if "reset_observation" in schema: - reset_schema_json = json.dumps( + reset_schema_payload = json.dumps( schema["reset_observation"], allow_nan=False, ensure_ascii=False, separators=(",", ":"), ) + if validation_token and validation_token in reset_schema_payload: + raise ValueError("schema contains validation credentials") + reset_schema_json = reset_schema_payload endpoint = urlsplit(base_url) ws_url = urlunsplit( diff --git a/tests/test_validation/test_runtime_collector.py b/tests/test_validation/test_runtime_collector.py index 742e50ea8b..53dd8b7ae0 100644 --- a/tests/test_validation/test_runtime_collector.py +++ b/tests/test_validation/test_runtime_collector.py @@ -112,14 +112,23 @@ def test_uncompressed_schema_still_obeys_total_byte_budget(monkeypatch, plan): assert evidence.observation_schema_json is None -def test_schema_cannot_persist_validation_credential(monkeypatch, plan): +@pytest.mark.parametrize("schema_key", ["observation", "reset_observation"]) +@pytest.mark.parametrize("escaped", [False, True]) +def test_schema_cannot_persist_validation_credential( + monkeypatch, plan, schema_key, escaped +): token = "validation-secret-value-" * 2 - stream = TrackedStream(json.dumps({"observation": {"description": token}}).encode()) + schemas = {"observation": {"type": "object"}} + schemas[schema_key] = {"properties": {token: {"description": token}}} + payload = json.dumps(schemas) + if escaped: + payload = payload.replace(token, "".join(f"\\u{ord(c):04x}" for c in token)) + stream = TrackedStream(payload.encode()) schema_transport(monkeypatch, stream) evidence = collector.collect_runtime_evidence( "http://127.0.0.1:8000", plan, episode_timeout_s=2, validation_token=token ) - assert evidence.observation_schema_json is None + assert getattr(evidence, f"{schema_key}_schema_json") is None assert evidence.failure_phase == "schema" assert token not in str(evidence) diff --git a/tests/validation_runtime/acceptance.json b/tests/validation_runtime/acceptance.json index 8144a29a5d..3c4cae5759 100644 --- a/tests/validation_runtime/acceptance.json +++ b/tests/validation_runtime/acceptance.json @@ -10,6 +10,8 @@ "tests.test_validation.integration.test_served_probe::test_failed_start_is_explicit", "tests.test_validation.integration.test_served_probe::test_real_echo_reset_schema_and_unscored_rewards[False]", "tests.test_validation.integration.test_served_probe::test_real_echo_reset_schema_and_unscored_rewards[True]", + "tests.test_validation.integration.test_session_telemetry_protocol::test_factory_error_survives_real_telemetry_handshake[False]", + "tests.test_validation.integration.test_session_telemetry_protocol::test_factory_error_survives_real_telemetry_handshake[True]", "tests.test_validation.integration.test_session_telemetry_protocol::test_same_session_seed_scores_and_subject_record[SessionEnv-True]", "tests.test_validation.integration.test_session_telemetry_protocol::test_same_session_seed_scores_and_subject_record[DropsSeed-False]", "tests.test_validation.integration.test_session_telemetry_protocol::test_same_session_seed_scores_and_subject_record[AsyncSeed-True]", From 2c9a48639d82d0e7b6f78d07d71946944a6577ae Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Fri, 2 Oct 2026 13:39:14 +0200 Subject: [PATCH 24/29] docs: explain runtime contracts --- docs/source/reference/cli.md | 12 +++++++++--- tests/validation_runtime/README.md | 10 +++++----- 2 files changed, 14 insertions(+), 8 deletions(-) diff --git a/docs/source/reference/cli.md b/docs/source/reference/cli.md index 59a7c31d59..3ef5c406bb 100644 --- a/docs/source/reference/cli.md +++ b/docs/source/reference/cli.md @@ -64,9 +64,15 @@ with dependent contract checks skipped and the completed trace retained. Validation does not inherit host credentials, and the CLI currently has no secret injection mechanism. Environments requiring a judge API key can therefore fail at -session creation. The observation check currently applies the advertised schema -to reset and step responses. Step rewards must be finite numbers, even though core -models allow null rewards; these are the current RFC 008 validation rules. +session creation. + +Observation validation uses the `reset_observation` field from `/schema` for resets +and `observation` for steps. Servers can declare `reset_observation_cls` when +reset returns a different observation type; it defaults to the step observation +class. Older servers without a reset schema use the step schema for both. +The Level 2 profile permits null rewards on reset and nonterminal steps. Terminal +steps require numeric rewards, and every numeric reward must be finite and within +the declared range; boolean rewards are invalid. Cleanup removes run-owned containers. Built images remain in Docker's local cache for reuse; the report records their immutable image IDs. Remove an unwanted image diff --git a/tests/validation_runtime/README.md b/tests/validation_runtime/README.md index ff6548b0a9..55353e667e 100644 --- a/tests/validation_runtime/README.md +++ b/tests/validation_runtime/README.md @@ -42,11 +42,11 @@ a unique image label for independent cleanup verification. The Echo canary copies the actual `envs/echo_env` sources unchanged and records their hashes. A test overlay adds only the execution declaration, replay plan and pinned offline image recipe. It runs `echo_message` and `echo_with_length` in one -session and verifies episode identity and state counts 0, 1, 2. Echo currently -returns null step rewards and a reset observation that lacks the advertised -`tool_name` field: the canary therefore expects explicit reward/schema **FAIL** -findings and CLI exit 1. Its passing test means those compatibility findings were -observed correctly; it does not mean Echo passed runtime validation. Inspect +session and verifies episode identity and state counts 0, 1, 2. Echo advertises a +separate reset-observation schema and emits null rewards on nonterminal steps, +which the Level 2 profile permits. The canary expects reward/schema **PASS** +findings and CLI exit 0; the remaining unimplemented checks still make the result +**WARN**, not complete Level 2 validation. Inspect `cli/echo_canary/compatibility-findings.json` for the actual results. Evidence is written to `outputs/validation-runtime//`, including source, From 270461550234982ddf69c1615876cf30ddbc7862 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Fri, 2 Oct 2026 13:43:22 +0200 Subject: [PATCH 25/29] fix: locate echo in isolated protocol tests --- tests/test_validation/integration/test_served_probe.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/test_validation/integration/test_served_probe.py b/tests/test_validation/integration/test_served_probe.py index 57eb591041..5f2c432ab2 100644 --- a/tests/test_validation/integration/test_served_probe.py +++ b/tests/test_validation/integration/test_served_probe.py @@ -73,7 +73,9 @@ def test_failed_start_is_explicit(): @pytest.mark.parametrize("legacy_schema", [False, True]) -def test_real_echo_reset_schema_and_unscored_rewards(legacy_schema): +def test_real_echo_reset_schema_and_unscored_rewards(legacy_schema, monkeypatch): + # The lab deliberately clears PYTHONPATH and tests the installed core wheel. + monkeypatch.syspath_prepend(str(Path(__file__).parents[3] / "envs")) from echo_env.server.echo_environment import EchoEnvironment from openenv.core.env_server.http_server import create_fastapi_app from openenv.core.env_server.mcp_types import CallToolAction, CallToolObservation From f60fa9e0d3c16afe61ec95d5d190174295a2e297 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Fri, 2 Oct 2026 13:49:46 +0200 Subject: [PATCH 26/29] fix: isolate factory test setup --- .../integration/test_session_telemetry_protocol.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_validation/integration/test_session_telemetry_protocol.py b/tests/test_validation/integration/test_session_telemetry_protocol.py index d61b75f023..924661ae80 100644 --- a/tests/test_validation/integration/test_session_telemetry_protocol.py +++ b/tests/test_validation/integration/test_session_telemetry_protocol.py @@ -85,6 +85,7 @@ def __init__(self): raise RuntimeError(TOKEN) monkeypatch.setenv("OPENENV_VALIDATION_TOKEN", TOKEN) + monkeypatch.setenv("ENABLE_WEB_INTERFACE", "false") app = create_app(BrokenFactory, ValueAction, Observation) server = uvicorn.Server(uvicorn.Config(app, log_level="critical")) with socket.socket() as listener: From 54736417161595bdcc9cc1c6cea8d650a0192fc3 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Fri, 2 Oct 2026 14:45:09 +0200 Subject: [PATCH 27/29] docs: clarify failure handling --- src/openenv/validation/runtime/collector.py | 1 + .../integration/test_session_telemetry_protocol.py | 1 + 2 files changed, 2 insertions(+) diff --git a/src/openenv/validation/runtime/collector.py b/src/openenv/validation/runtime/collector.py index dbc42c7e36..21b264fbd8 100644 --- a/src/openenv/validation/runtime/collector.py +++ b/src/openenv/validation/runtime/collector.py @@ -204,6 +204,7 @@ def receive_response(request): ): server_code = _server_error_code(json.loads(raw)) except Exception: + # Preserve the original ConnectionClosed if diagnostics fail. pass raise return connection.recv(timeout=remaining()) diff --git a/tests/test_validation/integration/test_session_telemetry_protocol.py b/tests/test_validation/integration/test_session_telemetry_protocol.py index 924661ae80..e20c30d3a9 100644 --- a/tests/test_validation/integration/test_session_telemetry_protocol.py +++ b/tests/test_validation/integration/test_session_telemetry_protocol.py @@ -82,6 +82,7 @@ def test_factory_error_survives_real_telemetry_handshake( ): class BrokenFactory(SessionEnv): def __init__(self): + # Fail before base initialization to exercise factory errors. raise RuntimeError(TOKEN) monkeypatch.setenv("OPENENV_VALIDATION_TOKEN", TOKEN) From d27048ea67d3e7192560857c1f26629ad4561af9 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Mon, 5 Oct 2026 13:47:17 +0200 Subject: [PATCH 28/29] fix: derive runtime checks from registry --- src/openenv/validation/graders/__init__.py | 4 ++ src/openenv/validation/runner.py | 74 +++++++++++++------- tests/test_validation/test_runner.py | 78 +++++++++++++++++++++- 3 files changed, 131 insertions(+), 25 deletions(-) diff --git a/src/openenv/validation/graders/__init__.py b/src/openenv/validation/graders/__init__.py index 26138e1a21..203ba00400 100644 --- a/src/openenv/validation/graders/__init__.py +++ b/src/openenv/validation/graders/__init__.py @@ -136,6 +136,10 @@ def load_entry_points(self) -> int: loaded += 1 return loaded + def get(self, check_id: str) -> Grader | None: + """Return the registered implementation, or `None` for a reserved check.""" + return self._graders.get(check_id) + def select(self, manifest: NormalizedManifest, max_level: Level) -> list[Grader]: """ Select the graders that apply to a manifest, up to a level ceiling. diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index 56f2891df0..06dfcec924 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -24,7 +24,7 @@ from .runtime.artifacts import write_runtime_bundle from .runtime.collector import collect_runtime_evidence, RuntimeCollectionInterrupted from .runtime.contracts import LaunchSpec, load_runtime_plan, RuntimePlanError -from .runtime.scheduler import execute_graders +from .runtime.scheduler import execute_graders, order_graders from .signature import detect_signature from .types import CheckStatus, Lane, Level, ProviderCapability @@ -84,6 +84,16 @@ def default_parser_registry() -> ParserRegistry: return registry +def default_grader_registry(policy: SeverityPolicy) -> GraderRegistry: + """The graders shipped in this build, including their selection metadata.""" + registry = GraderRegistry() + registry.register(StaticManifestGrader(policy.bounds)) + registry.register(RewardWellFormedGrader()) + registry.register(ObservationSchemaGrader()) + registry.register(StateContractGrader()) + return registry + + def _outcome(check_id, status, reason, *, started=None, measured=None): return CheckResult( check_id=check_id, @@ -106,7 +116,7 @@ def _applicable(check_id, manifest): return True -def _runtime(subject, *, skip_build, provider): +def _runtime(subject, graders, *, skip_build, provider): """Own build/start/collection/stop and retain failure evidence through teardown.""" manifest = subject.manifest plan = None @@ -181,11 +191,7 @@ def _runtime(subject, *, skip_build, provider): subject, image_ref=image_ref, running=running, runtime_evidence=evidence ) checks = execute_graders( - [ - RewardWellFormedGrader(), - ObservationSchemaGrader(), - StateContractGrader(), - ], + graders, subject, provider_capabilities=provider.capabilities, prior=[result], @@ -296,8 +302,10 @@ def run_validation( digest_before = "" parser = default_parser_registry().parser_for(signature) + graders = default_grader_registry(policy) manifest: NormalizedManifest | None = None results: list[CheckResult] = [] + selected = {} parse_started = time.monotonic() if not digest_before: @@ -329,8 +337,9 @@ def run_validation( inspection = {} cleanup = {"required": False, "completed": True} if manifest is not None: - graders = GraderRegistry() - graders.register(StaticManifestGrader(policy.bounds)) + selected = { + grader.check_id: grader for grader in graders.select(manifest, max_level) + } subject = Subject( root=target, manifest=manifest, @@ -338,10 +347,26 @@ def run_validation( running=None, outputs_dir=target / "outputs", ) - results.extend(execute_graders(graders.select(manifest, Level.STATIC), subject)) + results.extend( + execute_graders( + [ + grader + for grader in selected.values() + if grader.level is Level.STATIC + ], + subject, + ) + ) if wants_runtime: runtime_results, attempted, plan, evidence, inspection, cleanup = _runtime( - subject, skip_build=skip_build, provider=provider + subject, + [ + grader + for grader in selected.values() + if grader.level is Level.RUNTIME + ], + skip_build=skip_build, + provider=provider, ) results.extend(runtime_results) if attempted: @@ -352,21 +377,23 @@ def run_validation( # incomplete surface explicit throughout the staged implementation. present = {result.check_id for result in results} for entry in policy.entries_for_lane(Lane.LOCAL).values(): + grader = graders.get(entry.check_id) + applicable = ( + manifest is None + or entry.check_id in selected + or (grader is None and _applicable(entry.check_id, manifest)) + ) if ( entry.level in {Level.RUNTIME, Level.SEMANTIC} and entry.level <= max_level and entry.check_id not in present - and _applicable(entry.check_id, manifest) + and applicable ): reason = "grader not implemented in this build" if manifest is None: reason = "unmet dependency: valid manifest" - elif entry.check_id in { - "runtime.reward_well_formed", - "runtime.observation_schema", - "runtime.state_contract", - }: - reason = "unmet dependency: runtime.startup" + elif grader is not None and grader.depends_on: + reason = "unmet dependency: " + ", ".join(grader.depends_on) results.append(_outcome(entry.check_id, CheckStatus.SKIP, reason)) source_problem = None if not digest_before: @@ -383,18 +410,17 @@ def run_validation( else "package source could not be verified after validation" ) if source_problem is not None: + invalidated = {"runtime.startup"} + for grader in order_graders(list(selected.values())): + if invalidated.intersection(grader.depends_on): + invalidated.add(grader.check_id) results = [ _outcome( r.check_id, CheckStatus.SKIP, f"unmet dependency: runtime.startup ({source_problem})", ) - if r.check_id - in { - "runtime.reward_well_formed", - "runtime.observation_schema", - "runtime.state_contract", - } + if r.check_id in invalidated else r for r in results if r.check_id != "runtime.startup" diff --git a/tests/test_validation/test_runner.py b/tests/test_validation/test_runner.py index 87269c4e61..a484152534 100644 --- a/tests/test_validation/test_runner.py +++ b/tests/test_validation/test_runner.py @@ -4,11 +4,87 @@ import pytest from conftest import FIXTURES +from openenv.validation import runner from openenv.validation.policy import load_policy -from openenv.validation.report import ValidationReport +from openenv.validation.report import CheckResult, ValidationReport from openenv.validation.runner import run_validation, source_digest from openenv.validation.signature import SignatureError from openenv.validation.types import CheckStatus, Lane, Level, Verdict +from support.runtime import evidence, FakeRuntimeProvider + + +@pytest.mark.parametrize( + "mode", ["run", "skip-build", "source-change", "capability", "applies-to"] +) +def test_registered_runtime_graders_use_metadata_throughout_run( + tmp_path, monkeypatch, mode +): + package = tmp_path / "subject" + shutil.copytree(FIXTURES / "runtime" / "served_probe", package) + policy = load_policy("v2") + registry = runner.default_grader_registry(policy) + calls = [] + + class Probe: + level = Level.RUNTIME + requires_provider = frozenset() + requires_capabilities = ( + frozenset({"task_api"}) if mode == "capability" else frozenset() + ) + + def __init__(self, check_id, dependency): + self.check_id = check_id + self.depends_on = (dependency,) + + def applies_to(self, manifest): + return mode != "applies-to" + + def run(self, subject): + calls.append(self.check_id) + return CheckResult( + check_id=self.check_id, status=CheckStatus.PASS, duration_s=0 + ) + + # Register test implementations against existing policy IDs. The runner must + # discover them without adding their IDs to execution or dependency lists. + probes = [ + Probe("runtime.network_policy", "runtime.startup"), + Probe("runtime.host_containment", "runtime.network_policy"), + ] + for probe in probes: + registry.register(probe) + monkeypatch.setattr(runner, "default_grader_registry", lambda policy: registry) + + def collect(*args, **kwargs): + if mode == "source-change": + (package / "changed.txt").write_text("changed during collection") + return evidence() + + monkeypatch.setattr(runner, "collect_runtime_evidence", collect) + report = run_validation( + package, + max_level=Level.RUNTIME, + provider=FakeRuntimeProvider(), + skip_build=mode == "skip-build", + policy=policy, + ) + results = {result.check_id: result for result in report.results} + if mode in {"capability", "applies-to"}: + assert calls == [] + assert all(probe.check_id not in results for probe in probes) + return + assert calls == ([] if mode == "skip-build" else [p.check_id for p in probes]) + for probe in probes: + result = results[probe.check_id] + assert result.status is ( + CheckStatus.PASS if mode == "run" else CheckStatus.SKIP + ) + if mode == "skip-build": + assert result.evidence == ["unmet dependency: " + probe.depends_on[0]] + elif mode == "source-change": + assert result.evidence == [ + "unmet dependency: runtime.startup (package source changed during validation)" + ] def test_valid_package_passes_static_level(): From b9e3e29044d87383fc8db8ccfd4ac8d15b1f9809 Mon Sep 17 00:00:00 2001 From: burtenshaw Date: Mon, 5 Oct 2026 13:57:46 +0200 Subject: [PATCH 29/29] fix: enforce level two contracts --- docs/source/reference/cli.md | 24 +++-- rfcs/008-environment-auto-validation.md | 36 ++++++-- .../validation/graders/runtime/basic.py | 6 +- src/openenv/validation/manifest.py | 5 ++ src/openenv/validation/runner.py | 4 + .../schemas/manifest-v2.schema.json | 7 +- .../validation/schemas/report-v2.schema.json | 7 +- .../validation/runtime/echo_canary/README.md | 11 +-- .../integration/test_echo_canary.py | 10 +-- .../integration/test_served_probe.py | 2 +- .../test_validation/test_runtime_contracts.py | 23 +++++ .../test_validation/test_runtime_execution.py | 88 +++++++++++++++++++ tests/test_validation/test_runtime_grading.py | 42 +++++++-- tests/validation_runtime/README.md | 9 +- 14 files changed, 232 insertions(+), 42 deletions(-) diff --git a/docs/source/reference/cli.md b/docs/source/reference/cli.md index 3ef5c406bb..349b1f4d08 100644 --- a/docs/source/reference/cli.md +++ b/docs/source/reference/cli.md @@ -62,17 +62,27 @@ The declared `episode_timeout_s` bounds collection; a reset or judged step may u its remaining budget. Collection failures appear once under `runtime.startup`, with dependent contract checks skipped and the completed trace retained. -Validation does not inherit host credentials, and the CLI currently has no secret -injection mechanism. Environments requiring a judge API key can therefore fail at -session creation. +Credential delivery is deferred for this release. Set +`validation.execution.requires_credentials: true` when the environment needs an +externally supplied credential, such as a judge API key. This boolean declaration +causes a named `credential_delivery` SKIP before build or launch; dependent runtime +checks also SKIP. The resulting WARN does not establish complete Level 2 coverage. +The declaration accepts no secret values and defaults to false. Validation never +inherits host credentials or forwards API keys. Self-contained LLM judges can run +without this requirement; `llm_judged` alone does not imply a need for credentials. +An undeclared startup or reset failure still fails validation. Observation validation uses the `reset_observation` field from `/schema` for resets and `observation` for steps. Servers can declare `reset_observation_cls` when reset returns a different observation type; it defaults to the step observation -class. Older servers without a reset schema use the step schema for both. -The Level 2 profile permits null rewards on reset and nonterminal steps. Terminal -steps require numeric rewards, and every numeric reward must be finite and within -the declared range; boolean rewards are invalid. +class. Older servers without a reset schema use the step schema for both. Reset +validation is always required; the validator does not silently substitute a base +schema. These are output schemas, separate from any reset input declaration. +The Level 2 profile permits null rewards only on reset. Every step, including a +nonterminal step, requires a finite numeric reward within the declared range; +boolean rewards are invalid. This certification requirement is stricter than the +backward-compatible core `Observation.reward` type. Emit zero only when the intended +reward is zero; the validator never converts missing or null step rewards to zero. Cleanup removes run-owned containers. Built images remain in Docker's local cache for reuse; the report records their immutable image IDs. Remove an unwanted image diff --git a/rfcs/008-environment-auto-validation.md b/rfcs/008-environment-auto-validation.md index 43de87ee43..5989163f06 100644 --- a/rfcs/008-environment-auto-validation.md +++ b/rfcs/008-environment-auto-validation.md @@ -482,6 +482,7 @@ validation: dockerfile: Dockerfile context: . agent_boundary: api + requires_credentials: false ``` The other manifest sections remain authoritative for capabilities, resources, @@ -529,13 +530,20 @@ alongside the transcript; reconstruct observation plus the separate reward/done envelope before validating against the operation's schema. Neither schema may retrieve non-local references. -Reset reward may be null. A step with `done: false` may have null reward to denote -that no score was emitted. A terminal step must emit a finite JSON number within -the declared range. Every numeric reset or step reward is checked against that -range; booleans are invalid under this Level Two profile. This does not narrow the -core Observation model's backward-compatible reward type. State must retain the -requested episode identity, reset its step count to zero and advance it coherently -for successful steps. +Reset validation is required even when reset and step return different models. +An explicit reset output schema describes that difference; the validator never +silently substitutes the base `Observation` schema or skips reset. Reset input +declarations are a separate contract from these observation output schemas. + +Reset reward may be null. Every step, including a nonterminal step, must emit a +finite JSON number within the declared range. Every numeric reset reward is also +checked against that range; booleans are invalid. These are Level Two validation +requirements, stricter than the core API: the core `Observation.reward` type +continues to accept `bool | int | float | None` for compatibility. A valid core +environment can therefore fail Level Two validation. Authors should emit numeric +zero when a step's intended reward is zero; the validator never converts an absent +or null reward to zero. State must retain the requested episode identity, reset +its step count to zero and advance it coherently for successful steps. ### Startup, policy and provider supervision @@ -554,6 +562,20 @@ The provider owns build, readiness, bounded exec/logs, inspected settings and idempotent cleanup on success, failure, timeout and cancellation. Cleanup evidence must establish that no run-owned subject remains. Core provider ABCs are unchanged. +Credential delivery is deferred for this release. An environment that needs an +externally supplied credential declares `validation.execution.requires_credentials: +true` (a strict boolean, default `false`). This records a prerequisite only; no +credential names, values, sources or injection mechanism are accepted. After +validating the public plan, the runner reports `runtime.startup` as SKIP with the +missing `credential_delivery` capability named, before building or launching the +subject. Dependent checks remain SKIP and the report cannot establish complete +Level Two conformance. Missing credentials are never inferred from exception text +or from `llm_judged`: a self-contained judge may run without external credentials. +An undeclared startup or reset failure remains FAIL. Host tokens are never a +fallback, and credentials must not be embedded in manifests, plans or images. +Supporting external judge credentials requires a later explicit contract for +delivery, access isolation, lifetime, network access and evidence redaction. + Launches use no privilege, host namespaces, host-directory mounts, Docker socket or forwarded credentials. They drop Linux capabilities, enable no-new-privileges, use a read-only image and explicitly bounded writable roots, and expose only a diff --git a/src/openenv/validation/graders/runtime/basic.py b/src/openenv/validation/graders/runtime/basic.py index 79a95a33ac..2f63c76e60 100644 --- a/src/openenv/validation/graders/runtime/basic.py +++ b/src/openenv/validation/graders/runtime/basic.py @@ -77,7 +77,7 @@ def run(self, subject) -> CheckResult: class RewardWellFormedGrader(_RuntimeGrader): - """Allow unscored nonterminal steps; bound every emitted numeric reward.""" + """Require numeric step rewards for Level Two, beyond the core API contract.""" check_id = "runtime.reward_well_formed" @@ -97,9 +97,7 @@ def check(self, subject, evidence) -> list[str]: problems.append(f"exchange {index}: missing reward") continue reward = data["reward"] - if reward is None and ( - exchange.operation == "reset" or data.get("done") is False - ): + if reward is None and exchange.operation == "reset": continue if ( type(reward) not in (int, float) diff --git a/src/openenv/validation/manifest.py b/src/openenv/validation/manifest.py index de5a143330..a4623da3b6 100644 --- a/src/openenv/validation/manifest.py +++ b/src/openenv/validation/manifest.py @@ -327,6 +327,10 @@ class ExecutionDeclaration(BaseModel): Build context relative to the package root. agent_boundary (`str`): The access granted to an agent; currently only API access is supported. + requires_credentials (`bool`, *optional*, defaults to `False`): + Whether runtime execution needs credentials. Credential delivery is + deferred for this release; declaring `True` skips runtime validation. + This declaration accepts no credential names or values. """ model_config = ConfigDict(extra="forbid", frozen=True) @@ -336,6 +340,7 @@ class ExecutionDeclaration(BaseModel): dockerfile: str = "Dockerfile" context: str = "." agent_boundary: Literal["api"] = "api" + requires_credentials: bool = Field(default=False, strict=True) @field_validator("probe_path", "dockerfile", "context") @classmethod diff --git a/src/openenv/validation/runner.py b/src/openenv/validation/runner.py index 06dfcec924..842e4df68f 100644 --- a/src/openenv/validation/runner.py +++ b/src/openenv/validation/runner.py @@ -138,6 +138,10 @@ def _runtime(subject, graders, *, skip_build, provider): "missing validation.execution declaration and runtime plan" ) plan = load_runtime_plan(subject.root, manifest.execution) + if manifest.execution.requires_credentials: + raise UnsupportedCapability( + "credential_delivery is deferred for this release" + ) if provider is None: from .providers.docker import DockerValidationProvider diff --git a/src/openenv/validation/schemas/manifest-v2.schema.json b/src/openenv/validation/schemas/manifest-v2.schema.json index ab8bc785fa..93525bff6e 100644 --- a/src/openenv/validation/schemas/manifest-v2.schema.json +++ b/src/openenv/validation/schemas/manifest-v2.schema.json @@ -73,7 +73,7 @@ }, "ExecutionDeclaration": { "additionalProperties": false, - "description": "Author-declared runtime binding, introduced by manifest schema 2.\n\nAll paths are portable package-relative paths. Resolving the probe or build\ncontext must additionally reject symlink escapes before reading source files.\nThe data-only probe supplies actions; it cannot change declared capabilities.\n\nAttributes:\n kind (`str`):\n The implemented transport binding, `\"openenv_ws\"`.\n probe_path (`str`):\n JSON file containing the versioned runtime plan.\n dockerfile (`str`):\n Dockerfile relative to the package root.\n context (`str`):\n Build context relative to the package root.\n agent_boundary (`str`):\n The access granted to an agent; currently only API access is supported.", + "description": "Author-declared runtime binding, introduced by manifest schema 2.\n\nAll paths are portable package-relative paths. Resolving the probe or build\ncontext must additionally reject symlink escapes before reading source files.\nThe data-only probe supplies actions; it cannot change declared capabilities.\n\nAttributes:\n kind (`str`):\n The implemented transport binding, `\"openenv_ws\"`.\n probe_path (`str`):\n JSON file containing the versioned runtime plan.\n dockerfile (`str`):\n Dockerfile relative to the package root.\n context (`str`):\n Build context relative to the package root.\n agent_boundary (`str`):\n The access granted to an agent; currently only API access is supported.\n requires_credentials (`bool`, *optional*, defaults to `False`):\n Whether runtime execution needs credentials. Credential delivery is\n deferred for this release; declaring `True` skips runtime validation.\n This declaration accepts no credential names or values.", "properties": { "agent_boundary": { "const": "api", @@ -101,6 +101,11 @@ "default": "validation/runtime.json", "title": "Probe Path", "type": "string" + }, + "requires_credentials": { + "default": false, + "title": "Requires Credentials", + "type": "boolean" } }, "title": "ExecutionDeclaration", diff --git a/src/openenv/validation/schemas/report-v2.schema.json b/src/openenv/validation/schemas/report-v2.schema.json index d480354632..4f4d16e311 100644 --- a/src/openenv/validation/schemas/report-v2.schema.json +++ b/src/openenv/validation/schemas/report-v2.schema.json @@ -134,7 +134,7 @@ }, "ExecutionDeclaration": { "additionalProperties": false, - "description": "Author-declared runtime binding, introduced by manifest schema 2.\n\nAll paths are portable package-relative paths. Resolving the probe or build\ncontext must additionally reject symlink escapes before reading source files.\nThe data-only probe supplies actions; it cannot change declared capabilities.\n\nAttributes:\n kind (`str`):\n The implemented transport binding, `\"openenv_ws\"`.\n probe_path (`str`):\n JSON file containing the versioned runtime plan.\n dockerfile (`str`):\n Dockerfile relative to the package root.\n context (`str`):\n Build context relative to the package root.\n agent_boundary (`str`):\n The access granted to an agent; currently only API access is supported.", + "description": "Author-declared runtime binding, introduced by manifest schema 2.\n\nAll paths are portable package-relative paths. Resolving the probe or build\ncontext must additionally reject symlink escapes before reading source files.\nThe data-only probe supplies actions; it cannot change declared capabilities.\n\nAttributes:\n kind (`str`):\n The implemented transport binding, `\"openenv_ws\"`.\n probe_path (`str`):\n JSON file containing the versioned runtime plan.\n dockerfile (`str`):\n Dockerfile relative to the package root.\n context (`str`):\n Build context relative to the package root.\n agent_boundary (`str`):\n The access granted to an agent; currently only API access is supported.\n requires_credentials (`bool`, *optional*, defaults to `False`):\n Whether runtime execution needs credentials. Credential delivery is\n deferred for this release; declaring `True` skips runtime validation.\n This declaration accepts no credential names or values.", "properties": { "agent_boundary": { "const": "api", @@ -162,6 +162,11 @@ "default": "validation/runtime.json", "title": "Probe Path", "type": "string" + }, + "requires_credentials": { + "default": false, + "title": "Requires Credentials", + "type": "boolean" } }, "title": "ExecutionDeclaration", diff --git a/tests/fixtures/validation/runtime/echo_canary/README.md b/tests/fixtures/validation/runtime/echo_canary/README.md index 3db95588ca..5d5f4e526a 100644 --- a/tests/fixtures/validation/runtime/echo_canary/README.md +++ b/tests/fixtures/validation/runtime/echo_canary/README.md @@ -11,8 +11,9 @@ the subject COPY and entrypoint. The validator never imports Echo on the host. The acceptance test requires real tool responses and a continuing episode with state counts 0, 1 and 2. Echo explicitly advertises its base reset observation separately from CallToolObservation step responses. Its null rewards denote -unscored, nonterminal tool calls, permitted by the Level Two reward profile. -Startup, state, reward and observation schema checks must pass. Unimplemented -checks still produce SKIP and the report remains WARN; this does not certify Echo -or complete Level 2 validation. Terminal null rewards, malformed reset observations -and legacy single-schema mismatches remain failures in dedicated regressions. +unscored, nonterminal tool calls permitted by the core API but rejected by Level +Two's stricter numeric-step-reward requirement. Startup, state and observation +schema checks must pass; reward must FAIL, making the report FAIL with CLI exit 1. +This is an expected certification finding, not a broken canary or a reason to +coerce null to zero. Malformed reset observations and legacy single-schema +mismatches remain failures in dedicated regressions. diff --git a/tests/test_validation/integration/test_echo_canary.py b/tests/test_validation/integration/test_echo_canary.py index 054751cf2c..4030b050c2 100644 --- a/tests/test_validation/integration/test_echo_canary.py +++ b/tests/test_validation/integration/test_echo_canary.py @@ -34,7 +34,7 @@ def test_reference_echo_validates_reset_and_unscored_steps(tmp_path): expected = { "runtime.startup": "pass", "runtime.state_contract": "pass", - "runtime.reward_well_formed": "pass", + "runtime.reward_well_formed": "fail", "runtime.observation_schema": "pass", } (evidence_root / "cli/echo_canary/compatibility-findings.json").write_text( @@ -51,7 +51,7 @@ def test_reference_echo_validates_reset_and_unscored_steps(tmp_path): } for check_id in expected }, - "interpretation": "Implemented runtime checks pass; skipped checks keep Echo uncertified.", + "interpretation": "Echo is valid under the core API, but null step rewards fail Level Two validation.", }, indent=2, sort_keys=True, @@ -94,9 +94,9 @@ def test_reference_echo_validates_reset_and_unscored_steps(tmp_path): assert step["reward"] is None assert step["done"] is False - assert result.returncode == 0 - assert report["verdict"] == "warn" - assert checks["runtime.reward_well_formed"]["status"] == "pass" + assert result.returncode == 1 + assert report["verdict"] == "fail" + assert checks["runtime.reward_well_formed"]["status"] == "fail" assert checks["runtime.observation_schema"]["status"] == "pass" assert "tool_name" not in trace[0]["response_json"]["data"]["observation"] metadata = artifacts["collector-evidence.json"] diff --git a/tests/test_validation/integration/test_served_probe.py b/tests/test_validation/integration/test_served_probe.py index 5f2c432ab2..d9c07328dc 100644 --- a/tests/test_validation/integration/test_served_probe.py +++ b/tests/test_validation/integration/test_served_probe.py @@ -163,7 +163,7 @@ async def legacy_send(message): runtime_evidence=evidence, manifest=SimpleNamespace(reward=SimpleNamespace(range=(0, 1))), ) - assert RewardWellFormedGrader().run(subject).status.value == "pass" + assert RewardWellFormedGrader().run(subject).status.value == "fail" assert StateContractGrader().run(subject).status.value == "pass" assert ObservationSchemaGrader().run(subject).status.value == ( "fail" if legacy_schema else "pass" diff --git a/tests/test_validation/test_runtime_contracts.py b/tests/test_validation/test_runtime_contracts.py index 15541b0cba..812ac51275 100644 --- a/tests/test_validation/test_runtime_contracts.py +++ b/tests/test_validation/test_runtime_contracts.py @@ -61,6 +61,29 @@ def test_execution_paths_reject_nonportable_or_escaped_locations(field, path): ExecutionDeclaration(**{field: path}) +def test_execution_defaults_to_no_credential_requirement(): + assert ExecutionDeclaration().requires_credentials is False + + +@pytest.mark.parametrize("required", [False, True]) +def test_execution_accepts_explicit_credential_requirement(required): + assert ( + ExecutionDeclaration(requires_credentials=required).requires_credentials + is required + ) + + +@pytest.mark.parametrize("required", ["true", "false", 0, 1, None, [], {}]) +def test_execution_credential_requirement_requires_boolean(required): + with pytest.raises(ValidationError, match="valid boolean"): + ExecutionDeclaration(requires_credentials=required) + + +@pytest.mark.parametrize("model", [NormalizedManifest, ValidationReport]) +def test_credential_declaration_does_not_change_schema_one(model): + assert "requires_credentials" not in json.dumps(model.model_json_schema()) + + def test_plan_symlink_cannot_escape_package(tmp_path): package = tmp_path / "package" package.mkdir() diff --git a/tests/test_validation/test_runtime_execution.py b/tests/test_validation/test_runtime_execution.py index ce473b9c57..ca99939ab5 100644 --- a/tests/test_validation/test_runtime_execution.py +++ b/tests/test_validation/test_runtime_execution.py @@ -157,6 +157,94 @@ def test_unsupported_capability_is_refused_before_build(package, change): ) +@pytest.mark.parametrize("default_provider", [False, True]) +def test_declared_credentials_skip_before_provider_access( + package, monkeypatch, default_provider +): + import yaml + + source = package / "openenv.yaml" + data = yaml.safe_load(source.read_text()) + data["validation"]["execution"]["requires_credentials"] = True + source.write_text(yaml.safe_dump(data)) + monkeypatch.setenv("OPENAI_API_KEY", "host-credential-must-not-be-used") + provider = FakeRuntimeProvider() + + def unexpected_provider(): + raise AssertionError("credential deferral must precede provider creation") + + monkeypatch.setattr( + "openenv.validation.providers.docker.DockerValidationProvider", + unexpected_provider, + ) + report = run_validation( + package, + max_level=Level.RUNTIME, + provider=None if default_provider else provider, + ) + + assert not provider.builds and not provider.launches + assert report.verdict.value == "warn" + assert report.levels_run == [Level.STATIC] + runtime_results = { + r.check_id: r for r in report.results if r.check_id.startswith("runtime.") + } + assert all(r.status is CheckStatus.SKIP for r in runtime_results.values()) + assert runtime_results["runtime.startup"].evidence == [ + "credential_delivery is deferred for this release" + ] + for name in ("reward_well_formed", "observation_schema", "state_contract"): + assert "runtime.startup" in runtime_results[f"runtime.{name}"].evidence[0] + assert "host-credential-must-not-be-used" not in report.model_dump_json() + + +def test_credential_deferral_does_not_hide_invalid_runtime_plan(package): + import yaml + + source = package / "openenv.yaml" + data = yaml.safe_load(source.read_text()) + data["validation"]["execution"]["requires_credentials"] = True + source.write_text(yaml.safe_dump(data)) + (package / "validation/runtime.json").write_text('{"actions": []}') + provider = FakeRuntimeProvider() + + report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + + assert report.verdict.value == "fail" + result = next(r for r in report.results if r.check_id == "runtime.startup") + assert result.status is CheckStatus.FAIL + assert "runtime plan schema validation failed" in result.evidence[0] + assert not provider.builds and not provider.launches + + +@pytest.mark.parametrize("declare_no_credentials", [False, True]) +def test_llm_judge_without_credentials_can_run( + package, monkeypatch, declare_no_credentials +): + import yaml + + source = package / "openenv.yaml" + data = yaml.safe_load(source.read_text()) + data["validation"]["capabilities"]["llm_judged"] = True + data["validation"]["reward"]["variance_tolerance"] = 0.1 + data["validation"]["judge"] = {"model": "offline-judge", "version": "1"} + if declare_no_credentials: + data["validation"]["execution"]["requires_credentials"] = False + source.write_text(yaml.safe_dump(data)) + provider = FakeRuntimeProvider() + monkeypatch.setattr( + "openenv.validation.runner.collect_runtime_evidence", + lambda *args, **kwargs: measured_episode(), + ) + + report = run_validation(package, max_level=Level.RUNTIME, provider=provider) + + assert len(provider.builds) == len(provider.launches) == 1 + result = next(r for r in report.results if r.check_id == "runtime.startup") + assert result.status is CheckStatus.PASS + assert provider.subject.stopped + + def test_explicit_v1_rejected_before_runtime(package): provider = FakeRuntimeProvider() with pytest.raises(PolicyError, match="policy v2"): diff --git a/tests/test_validation/test_runtime_grading.py b/tests/test_validation/test_runtime_grading.py index 3c8dba8591..bdd2f00744 100644 --- a/tests/test_validation/test_runtime_grading.py +++ b/tests/test_validation/test_runtime_grading.py @@ -6,6 +6,7 @@ import pytest from conftest import load_fixture_manifest +from openenv.core.env_server.types import Observation from openenv.validation.graders import Subject from openenv.validation.graders.runtime import ( basic, @@ -122,16 +123,39 @@ def test_step_rewards_are_checked_without_coercion(tmp_path, reward, done): assert any("reward" in message for message in result.evidence) -@pytest.mark.parametrize("done", [False, True, None, 0]) -def test_null_step_reward_requires_explicit_nonterminal_observation(tmp_path, done): +@pytest.mark.parametrize("reward", [None, True, False]) +@pytest.mark.parametrize("done", [False, True]) +def test_core_reward_types_do_not_imply_level_two_certification(tmp_path, reward, done): + # Core compatibility does not weaken the stricter certification contract. + assert Observation(reward=reward, done=done).reward is reward rows = mutate_response( - good_rows(), 2, lambda response: response["data"].update(reward=None, done=done) + good_rows(), + 2, + lambda response: response["data"].update(reward=reward, done=done), ) result = RewardWellFormedGrader().run(subject_with(tmp_path, rows)) - assert result.status is (CheckStatus.PASS if done is False else CheckStatus.FAIL) + assert result.status is CheckStatus.FAIL + assert any("exchange 2: reward" in message for message in result.evidence) + assert json.loads(rows[2].response_json)["data"]["reward"] is reward -def test_explicit_reset_schema_is_independent_of_step_schema(tmp_path): +def test_null_reset_reward_remains_unscored(tmp_path): + rows = good_rows() + assert json.loads(rows[0].response_json)["data"]["reward"] is None + assert ( + RewardWellFormedGrader().run(subject_with(tmp_path, rows)).status + is CheckStatus.PASS + ) + assert json.loads(rows[0].response_json)["data"]["reward"] is None + + +@pytest.mark.parametrize( + "invalid_index,invalid_observation", + [(0, {"ready": "yes"}), (2, {"counter": "not-an-integer"})], +) +def test_explicit_reset_schema_is_independent_of_step_schema( + tmp_path, invalid_index, invalid_observation +): rows = mutate_response( good_rows(), 0, @@ -155,13 +179,17 @@ def test_explicit_reset_schema_is_independent_of_step_schema(tmp_path): ) assert ObservationSchemaGrader().run(subject).status is CheckStatus.PASS malformed = mutate_response( - rows, 0, lambda response: response["data"].update(observation={"ready": "yes"}) + rows, + invalid_index, + lambda response: response["data"].update(observation=invalid_observation), ) subject = replace( subject, runtime_evidence=replace(subject.runtime_evidence, exchanges=tuple(malformed)), ) - assert ObservationSchemaGrader().run(subject).status is CheckStatus.FAIL + result = ObservationSchemaGrader().run(subject) + assert result.status is CheckStatus.FAIL + assert all(f"exchange {invalid_index}:" in message for message in result.evidence) @pytest.mark.parametrize( diff --git a/tests/validation_runtime/README.md b/tests/validation_runtime/README.md index 55353e667e..6803bb4e77 100644 --- a/tests/validation_runtime/README.md +++ b/tests/validation_runtime/README.md @@ -43,10 +43,11 @@ The Echo canary copies the actual `envs/echo_env` sources unchanged and records their hashes. A test overlay adds only the execution declaration, replay plan and pinned offline image recipe. It runs `echo_message` and `echo_with_length` in one session and verifies episode identity and state counts 0, 1, 2. Echo advertises a -separate reset-observation schema and emits null rewards on nonterminal steps, -which the Level 2 profile permits. The canary expects reward/schema **PASS** -findings and CLI exit 0; the remaining unimplemented checks still make the result -**WARN**, not complete Level 2 validation. Inspect +separate reset-observation schema and emits null rewards on nonterminal steps. +Those rewards remain valid under the core API but fail Level Two's stricter +numeric-step-reward requirement. The canary expects schema **PASS**, reward +**FAIL**, report **FAIL** and CLI exit 1. This demonstrates the certification +boundary without changing Echo or coercing its rewards. Inspect `cli/echo_canary/compatibility-findings.json` for the actual results. Evidence is written to `outputs/validation-runtime//`, including source,