From 572ed693117c14a60dd03ab1a4a0ce8255d24a11 Mon Sep 17 00:00:00 2001 From: "Mars.P" Date: Sat, 19 Sep 2026 07:38:12 +0800 Subject: [PATCH 1/4] Reclaim log-stream bytes nothing cites any more MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The artifact plane was the only one with a retention rule and a reaper. An agent run's log streams, their segments and the routed CAS objects behind them had neither: once a run ended, its archive sat at its destination for good, and nothing in the schema could even answer whether something still needed it. Two things were missing, and the first is why the second is safe. A sealed paired-qualification result is a statement ABOUT a set of runs, expressed nowhere as a foreign key — so "is this stream still cited" was unanswerable, and an unanswerable question has to mean keep. The pin table answers it, and is written in the seal's own transaction so a committed seal always has its pins. Migration 0235 backfills the same four closures for results sealed before it existed, because a pin table that only fills forward would make every older result read as citing nothing. The reaper then reclaims only what both waits and a fail-closed citation probe have cleared, bytes first and the tombstone last. The head row deliberately outlives its bytes: a reader that finds nothing cannot tell "reclaimed by policy" from "capture lost it", so the Room reads Purged instead of meeting a stream that will not load. --- ...ReapExpiredDurableRecordsCommandHandler.cs | 20 + .../DurableRetentionReaperRecurringJob.cs | 25 + .../Persistence/Db/CodeSpaceDbContext.cs | 1 + .../DbUpFiles/0235_durable_retention_pins.sql | 114 ++++ .../0236_agent_run_log_stream_retention.sql | 292 ++++++++++ .../Persistence/Entities/AgentRunLogStream.cs | 14 + .../Entities/PairedQualificationResultPin.cs | 46 ++ ...iredQualificationResultPinConfiguration.cs | 27 + .../PairedQualificationResultStore.cs | 30 + .../Services/Sessions/Room/RoomNarrative.cs | 9 +- .../Services/Sessions/Room/RoomProjector.cs | 32 +- .../Retention/ArtifactReferenceOracle.cs | 1 + .../Retention/DurableRetentionDecision.cs | 67 +++ .../Retention/DurableRetentionPolicy.cs | 53 ++ .../Retention/DurableRetentionReaper.cs | 138 +++++ .../Retention/IDurableRetentionCursor.cs | 53 ++ .../Retention/IDurableRetentionReaper.cs | 15 + .../ReapExpiredDurableRecordsCommand.cs | 24 + .../Dtos/Sessions/Room/RoomAgentLogStatus.cs | 8 + .../Retention/DurableRetention.cs | 60 ++ .../DurableRetentionReaperFlowTests.cs | 547 ++++++++++++++++++ .../RequestAuthorizationInventoryTests.cs | 1 + .../Sessions/Room/RoomLogSummaryTests.cs | 30 +- .../Artifacts/ArtifactReferenceOracleTests.cs | 1 + .../Retention/DurableRetentionPolicyTests.cs | 117 ++++ 25 files changed, 1711 insertions(+), 14 deletions(-) create mode 100644 backend/src/CodeSpace.Core/Handlers/CommandHandlers/Workflows/ReapExpiredDurableRecordsCommandHandler.cs create mode 100644 backend/src/CodeSpace.Core/Jobs/RecurringJobs/DurableRetentionReaperRecurringJob.cs create mode 100644 backend/src/CodeSpace.Core/Persistence/DbUpFiles/0235_durable_retention_pins.sql create mode 100644 backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql create mode 100644 backend/src/CodeSpace.Core/Persistence/Entities/PairedQualificationResultPin.cs create mode 100644 backend/src/CodeSpace.Core/Persistence/EntityConfigurations/PairedQualificationResultPinConfiguration.cs create mode 100644 backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs create mode 100644 backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs create mode 100644 backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs create mode 100644 backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs create mode 100644 backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionReaper.cs create mode 100644 backend/src/CodeSpace.Messages/Commands/Workflows/ReapExpiredDurableRecordsCommand.cs create mode 100644 backend/src/CodeSpace.Messages/Retention/DurableRetention.cs create mode 100644 backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs create mode 100644 backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs diff --git a/backend/src/CodeSpace.Core/Handlers/CommandHandlers/Workflows/ReapExpiredDurableRecordsCommandHandler.cs b/backend/src/CodeSpace.Core/Handlers/CommandHandlers/Workflows/ReapExpiredDurableRecordsCommandHandler.cs new file mode 100644 index 000000000..5132c4c7b --- /dev/null +++ b/backend/src/CodeSpace.Core/Handlers/CommandHandlers/Workflows/ReapExpiredDurableRecordsCommandHandler.cs @@ -0,0 +1,20 @@ +using CodeSpace.Core.Services.Workflows.Retention; +using CodeSpace.Messages.Commands.Workflows; +using MediatR; + +namespace CodeSpace.Core.Handlers.CommandHandlers.Workflows; + +/// Rule 16 — thin handler. The bounded sweep lives in . +public sealed class ReapExpiredDurableRecordsCommandHandler : IRequestHandler +{ + private readonly IDurableRetentionReaper _reaper; + + public ReapExpiredDurableRecordsCommandHandler(IDurableRetentionReaper reaper) { _reaper = reaper; } + + public async Task Handle(ReapExpiredDurableRecordsCommand request, CancellationToken cancellationToken) + { + var summary = await _reaper.SweepAsync(cancellationToken).ConfigureAwait(false); + + return new ReapExpiredDurableRecordsResponse { Summary = summary }; + } +} diff --git a/backend/src/CodeSpace.Core/Jobs/RecurringJobs/DurableRetentionReaperRecurringJob.cs b/backend/src/CodeSpace.Core/Jobs/RecurringJobs/DurableRetentionReaperRecurringJob.cs new file mode 100644 index 000000000..5ba0173ed --- /dev/null +++ b/backend/src/CodeSpace.Core/Jobs/RecurringJobs/DurableRetentionReaperRecurringJob.cs @@ -0,0 +1,25 @@ +using CodeSpace.Messages.Commands.Workflows; +using MediatR; + +namespace CodeSpace.Core.Jobs.RecurringJobs; + +/// +/// Hourly, dispatches to reclaim durable records nothing cites any +/// more. Thin Mediator dispatcher (Rule 14) — the work lives in +/// . +/// +/// Hourly is ample and deliberately unhurried: the shortest rule keeps a record for days before it is even a +/// candidate and then quarantines it for another day, so the cadence changes nothing about WHAT is reclaimed — only +/// how promptly. Offset to :45 so it does not pile onto the :15 artifact reaper or the :30 spool reaper. +/// +public sealed class DurableRetentionReaperRecurringJob : IRecurringJob +{ + private readonly IMediator _mediator; + + public DurableRetentionReaperRecurringJob(IMediator mediator) { _mediator = mediator; } + + public string JobId => nameof(DurableRetentionReaperRecurringJob); + public string CronExpression => "45 * * * *"; // quarter to every hour + + public async Task Execute() => await _mediator.Send(new ReapExpiredDurableRecordsCommand()).ConfigureAwait(false); +} diff --git a/backend/src/CodeSpace.Core/Persistence/Db/CodeSpaceDbContext.cs b/backend/src/CodeSpace.Core/Persistence/Db/CodeSpaceDbContext.cs index bc64aed63..bae1ef5c0 100644 --- a/backend/src/CodeSpace.Core/Persistence/Db/CodeSpaceDbContext.cs +++ b/backend/src/CodeSpace.Core/Persistence/Db/CodeSpaceDbContext.cs @@ -117,6 +117,7 @@ public CodeSpaceDbContext(DbContextOptions options, ICurrent public DbSet PairedQualificationCellAdmission => Set(); public DbSet PairedQualificationCellCheckpoint => Set(); public DbSet PairedQualificationResult => Set(); + public DbSet PairedQualificationResultPin => Set(); public DbSet Lesson => Set(); public DbSet SupervisorDecisionRecord => Set(); diff --git a/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0235_durable_retention_pins.sql b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0235_durable_retention_pins.sql new file mode 100644 index 000000000..66fcdf7f4 --- /dev/null +++ b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0235_durable_retention_pins.sql @@ -0,0 +1,114 @@ +-- 0235_durable_retention_pins.sql +-- +-- Two things nothing in the schema could say before: which durable records a sealed qualification result still cites, +-- and when a log stream's bytes became reclaimable. +-- +-- WHY A PIN TABLE. A qualification result is an immutable statement ABOUT a set of agent runs — their logs, their +-- cleanup receipts, their offloaded event payloads. None of that is expressed as a foreign key anywhere: the result row +-- carries digests and a JSON outcome, and the observations it was computed from name their run in benchmark_result. +-- So "is this log stream still cited by a sealed result" was a question no reaper could ask, and a reaper that cannot +-- ask it must never delete. The pin is the answer, written in the SAME transaction as the seal: if the seal commits, +-- every record it cites is pinned, and if it does not, no pin exists either. +-- +-- WHY TWO TARGET COLUMNS. ArtifactReferenceOracle enumerates every column in the schema whose name ends in +-- `artifact_id` and whose target is workflow_artifact — that naming rule IS its completeness argument. An artifact pin +-- must therefore live in a column called pinned_artifact_id and nothing else may, or the oracle would read a log +-- stream id as an artifact reference. Every other pinned kind uses pinned_id. The CHECK ties the two together so the +-- pair can never disagree with the kind. +-- +-- The uniqueness the pin needs is over (result, kind, target), which spans both nullable columns, so it is an +-- expression index rather than the primary key; the surrogate id keeps the row addressable like every other entity. +-- +-- WHY retain_until AND purged_at. A log stream's bytes live in routed CAS objects that nothing ever reclaimed. The +-- reaper's two waits need somewhere durable to stand: retain_until is the instant the bytes become reclaimable — the +-- reaper writes it on the FIRST sweep that finds the stream terminal, past its rule and cited by nobody, and collects +-- only on a later sweep once it has passed. purged_at is the tombstone. The stream head row deliberately SURVIVES its +-- bytes: a reader that finds no row cannot tell "purged by policy" from "lost", and the whole point of a retention +-- plane is that reclamation is legible. Both are metadata-only nullable ADD COLUMNs (no rewrite, no index), and NULL +-- is the correct reading for every row an older binary writes or reads. +-- +-- The rules that keep these two columns honest are in the guard function, not in a CHECK: see +-- 0236_agent_run_log_stream_retention.sql. A CHECK here would have to be validated against the whole table. +-- +-- Rollback: DROP TABLE paired_qualification_result_pin; ALTER TABLE agent_run_log_stream DROP COLUMN retain_until, +-- DROP COLUMN purged_at. Nothing reads either from an older binary. + +CREATE TABLE paired_qualification_result_pin ( + id uuid PRIMARY KEY, + result_id uuid NOT NULL REFERENCES paired_qualification_result(observation_group_id) ON DELETE CASCADE, + kind varchar(32) NOT NULL, + pinned_id uuid NULL, + pinned_artifact_id uuid NULL, + pinned_at timestamptz NOT NULL, + CONSTRAINT ck_paired_qualification_result_pin_kind + CHECK (kind IN ('LogStream', 'Artifact', 'CleanupReceipt', 'AgentRun')), + CONSTRAINT ck_paired_qualification_result_pin_target + CHECK (num_nonnulls(pinned_id, pinned_artifact_id) = 1 AND (kind = 'Artifact') = (pinned_artifact_id IS NOT NULL)) +); + +CREATE UNIQUE INDEX ux_paired_qualification_result_pin + ON paired_qualification_result_pin (result_id, kind, COALESCE(pinned_id, pinned_artifact_id)); + +-- One index per probed column, partial so each is only the rows that column addresses: every consumer asks +-- "EXISTS (SELECT 1 ... WHERE = $1)", and `= $1` implies NOT NULL, so the partial index serves it. +CREATE INDEX ix_paired_qualification_result_pin_target + ON paired_qualification_result_pin (pinned_id) WHERE pinned_id IS NOT NULL; +CREATE INDEX ix_paired_qualification_result_pin_artifact + ON paired_qualification_result_pin (pinned_artifact_id) WHERE pinned_artifact_id IS NOT NULL; + +COMMENT ON TABLE paired_qualification_result_pin IS + 'Which durable records a sealed paired-qualification result cites. Written in the seal''s own transaction; read by every retention cursor as the "still referenced" proof.'; + +-- Backfill, and it is not optional. A pin table that only fills going forward would make every result sealed before +-- this migration read as citing NOTHING, and the first reaper sweep would then find its evidence collectable. The four +-- statements below are the same four closures PairedQualificationResultStore writes for a new seal, in SQL: the runs a +-- result's observations name, and then that run's log streams, cleanup receipts and offloaded event payloads. They and +-- the writer are pinned against each other by an integration test. + +INSERT INTO paired_qualification_result_pin (id, result_id, kind, pinned_id, pinned_artifact_id, pinned_at) +SELECT gen_random_uuid(), cited.result_id, 'AgentRun', cited.agent_run_id, NULL, now() +FROM ( + SELECT DISTINCT result.observation_group_id AS result_id, observation.agent_run_id + FROM paired_qualification_result result + JOIN benchmark_result observation ON observation.observation_group_id = result.observation_group_id + WHERE observation.agent_run_id IS NOT NULL +) AS cited +ON CONFLICT DO NOTHING; + +INSERT INTO paired_qualification_result_pin (id, result_id, kind, pinned_id, pinned_artifact_id, pinned_at) +SELECT gen_random_uuid(), cited.result_id, 'LogStream', cited.stream_id, NULL, now() +FROM ( + SELECT DISTINCT pin.result_id, stream.id AS stream_id + FROM paired_qualification_result_pin pin + JOIN agent_run_log_stream stream ON stream.agent_run_id = pin.pinned_id + WHERE pin.kind = 'AgentRun' +) AS cited +ON CONFLICT DO NOTHING; + +INSERT INTO paired_qualification_result_pin (id, result_id, kind, pinned_id, pinned_artifact_id, pinned_at) +SELECT gen_random_uuid(), cited.result_id, 'CleanupReceipt', cited.receipt_id, NULL, now() +FROM ( + SELECT DISTINCT pin.result_id, receipt.id AS receipt_id + FROM paired_qualification_result_pin pin + JOIN agent_run_cleanup_receipt receipt ON receipt.agent_run_id = pin.pinned_id + WHERE pin.kind = 'AgentRun' +) AS cited +ON CONFLICT DO NOTHING; + +INSERT INTO paired_qualification_result_pin (id, result_id, kind, pinned_id, pinned_artifact_id, pinned_at) +SELECT gen_random_uuid(), cited.result_id, 'Artifact', NULL, cited.data_artifact_id, now() +FROM ( + SELECT DISTINCT pin.result_id, event.data_artifact_id + FROM paired_qualification_result_pin pin + JOIN agent_run_event event ON event.agent_run_id = pin.pinned_id + WHERE pin.kind = 'AgentRun' AND event.data_artifact_id IS NOT NULL +) AS cited +ON CONFLICT DO NOTHING; + +ALTER TABLE agent_run_log_stream ADD COLUMN retain_until timestamptz NULL; +ALTER TABLE agent_run_log_stream ADD COLUMN purged_at timestamptz NULL; + +COMMENT ON COLUMN agent_run_log_stream.retain_until IS + 'The earliest instant this stream''s bytes may be reclaimed, written on the first sweep that found it collectable. NULL means no sweep has ever proposed it.'; +COMMENT ON COLUMN agent_run_log_stream.purged_at IS + 'When this stream''s segment bytes were reclaimed. The head row survives as the tombstone so a reader reads "purged", never a missing stream.'; diff --git a/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql new file mode 100644 index 000000000..36c30d8a0 --- /dev/null +++ b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql @@ -0,0 +1,292 @@ +-- 0236_agent_run_log_stream_retention.sql +-- +-- A terminal log stream could not be told that its bytes are gone. +-- +-- agent_run_log_stream_guard() admits five update shapes (a capture claim, a source finalization, a remote-stall +-- statement, an exact one-segment head advance, a terminal transition) and refuses everything else outright: +-- +-- IF OLD.state <> 'Open' THEN RAISE EXCEPTION 'agent_run_log_stream terminal state is immutable (id=%).' ... +-- +-- That is right for every statement a capturing worker makes, and it is kept for all of them. But it also means the +-- retention columns 0235 added can never be written: a stream is a retention candidate only once it is terminal, and +-- from that moment the row admits no update at all. DELETE is refused too, and deliberately so — the head row is the +-- tombstone that lets a reader say "purged" instead of finding nothing. +-- +-- This migration adds ONE admissible shape, the sixth, and makes it as narrow as the row allows: a RETENTION +-- STATEMENT may move retain_until and purged_at on a terminal stream and NOTHING else. It cannot change the state, the +-- claim, the byte head, the source offsets, the finalization receipt, the digests or the error — so a purge can never +-- be mistaken for a capture verdict, and the durable prefix's metadata stays exactly as readable as it was. Three +-- further rules ride with it, each of which exists because its absence is silent data loss: +-- +-- * A stream that is still Open is never a retention candidate. Its bytes belong to a live capture session, so the +-- columns are refused there rather than left to fall through an arm that does not enumerate them. +-- * A purge requires a retain_until that was already recorded. The reaper's two waits are an age floor and then a +-- quarantine, and the second one only exists if it was durably written first; without this rule one sweep could +-- both propose and execute a collection. +-- * A purge is final. Clearing purged_at would claim bytes are back that no one restored. +-- +-- The INSERT arm is tightened for the same reason it rejects a pre-set manifest receipt: a new stream that arrives +-- already carrying a retention verdict is not a new stream. +-- +-- Supersedes 0232_agent_run_log_stream_owner_loss.sql, which is reproduced verbatim apart from the arm above and the +-- INSERT clause. The function has been redefined at 0129, 0132, 0133, 0201, 0230 and 0232 — one number, one +-- definition, every time (MigrationDiscoveryTests.No_two_migrations_at_one_number_redefine_the_same_function), because +-- DbUp is FILENAME-keyed: two files at one number both run and the later NAME silently wins. +-- +-- No DDL, no new column, no new trigger. +-- +-- Rollback: re-run 0232_agent_run_log_stream_owner_loss.sql's CREATE OR REPLACE FUNCTION body. Any stream already +-- carrying purged_at keeps it; nothing can write another one. + +CREATE OR REPLACE FUNCTION agent_run_log_stream_guard() RETURNS trigger AS $$ +DECLARE + appended agent_run_log_segment%ROWTYPE; + current_fence BIGINT; + is_claim BOOLEAN := FALSE; + is_source_finalize BOOLEAN := FALSE; + is_remote_stall BOOLEAN := FALSE; + is_owner_loss BOOLEAN := FALSE; +BEGIN + IF TG_OP = 'DELETE' THEN + RAISE EXCEPTION 'agent_run_log_stream is durable capture state — DELETE rejected (id=%).', OLD.id; + END IF; + + IF TG_OP = 'INSERT' THEN + IF NEW.manifest_digest IS NOT NULL THEN + RAISE EXCEPTION 'A new log stream cannot carry a manifest receipt.'; + END IF; + SELECT fence_epoch INTO current_fence FROM agent_run + WHERE team_id = NEW.team_id AND id = NEW.agent_run_id + FOR SHARE; + IF NOT FOUND OR current_fence <= 0 OR NEW.worker_fence_epoch IS DISTINCT FROM current_fence + OR NEW.capture_session_id IS NULL OR NEW.schema_version <> 3 THEN + RAISE EXCEPTION 'agent_run_log_stream requires the current positive AgentRun fence and a capture session (run_id=%, attempted_fence=%).', NEW.agent_run_id, NEW.worker_fence_epoch; + END IF; + IF NEW.state <> 'Open' OR NEW.revision <> 1 OR NEW.segment_count <> 0 OR NEW.total_bytes <> 0 + OR NEW.source_offset_bytes <> 0 OR NEW.capture_source_base_offset_bytes <> 0 + OR NEW.capture_finalized_at IS NOT NULL OR NEW.next_segment_ordinal <> 1 OR NEW.next_offset_bytes <> 0 + OR NEW.completed_at IS NOT NULL OR NEW.error_code IS NOT NULL OR NEW.error_message IS NOT NULL + OR NEW.content_digest IS NOT NULL OR NEW.content_digest_algorithm IS NOT NULL + OR NEW.remote_stall_since IS NOT NULL OR NEW.remote_stall_code IS NOT NULL + OR NEW.retain_until IS NOT NULL OR NEW.purged_at IS NOT NULL + OR NEW.last_modified_at < NEW.created_at THEN + RAISE EXCEPTION 'agent_run_log_stream must start as an empty Open revision-one head (id=%).', NEW.id; + END IF; + RETURN NEW; + END IF; + + -- The SIXTH admissible update shape, and the reason this function is redefined: a RETENTION STATEMENT. It is + -- matched first and returns on its own, because the arms below are written for statements a capturing worker + -- makes and every one of them refuses a terminal row. Both columns are named explicitly, so a statement that + -- touches neither never reaches this arm at all. + IF NEW.retain_until IS DISTINCT FROM OLD.retain_until OR NEW.purged_at IS DISTINCT FROM OLD.purged_at THEN + IF OLD.state = 'Open' THEN + RAISE EXCEPTION 'agent_run_log_stream retention statement rejected on a live stream (id=%); its bytes belong to an open capture session.', OLD.id; + END IF; + IF OLD.purged_at IS NOT NULL THEN + RAISE EXCEPTION 'agent_run_log_stream purge is final; a purged stream admits no further retention statement (id=%).', OLD.id; + END IF; + IF NEW.purged_at IS NOT NULL AND NEW.retain_until IS NULL THEN + RAISE EXCEPTION 'agent_run_log_stream cannot be purged without the retain_until it was quarantined under (id=%).', OLD.id; + END IF; + IF NEW.id IS DISTINCT FROM OLD.id OR NEW.team_id IS DISTINCT FROM OLD.team_id + OR NEW.agent_run_id IS DISTINCT FROM OLD.agent_run_id OR NEW.state IS DISTINCT FROM OLD.state + OR NEW.stream_kind IS DISTINCT FROM OLD.stream_kind OR NEW.content_type IS DISTINCT FROM OLD.content_type + OR NEW.content_encoding IS DISTINCT FROM OLD.content_encoding OR NEW.capture_source IS DISTINCT FROM OLD.capture_source + OR NEW.schema_version IS DISTINCT FROM OLD.schema_version OR NEW.retention IS DISTINCT FROM OLD.retention + OR NEW.expires_at IS DISTINCT FROM OLD.expires_at OR NEW.created_at IS DISTINCT FROM OLD.created_at + OR NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id + OR NEW.segment_count IS DISTINCT FROM OLD.segment_count OR NEW.total_bytes IS DISTINCT FROM OLD.total_bytes + OR NEW.source_offset_bytes IS DISTINCT FROM OLD.source_offset_bytes + OR NEW.capture_source_base_offset_bytes IS DISTINCT FROM OLD.capture_source_base_offset_bytes + OR NEW.next_segment_ordinal IS DISTINCT FROM OLD.next_segment_ordinal + OR NEW.next_offset_bytes IS DISTINCT FROM OLD.next_offset_bytes + OR NEW.capture_finalized_at IS DISTINCT FROM OLD.capture_finalized_at + OR NEW.completed_at IS DISTINCT FROM OLD.completed_at + OR NEW.error_code IS DISTINCT FROM OLD.error_code OR NEW.error_message IS DISTINCT FROM OLD.error_message + OR NEW.content_digest_algorithm IS DISTINCT FROM OLD.content_digest_algorithm + OR NEW.content_digest IS DISTINCT FROM OLD.content_digest + OR NEW.manifest_digest IS DISTINCT FROM OLD.manifest_digest + OR NEW.remote_stall_since IS DISTINCT FROM OLD.remote_stall_since + OR NEW.remote_stall_code IS DISTINCT FROM OLD.remote_stall_code THEN + RAISE EXCEPTION 'agent_run_log_stream retention statement cannot rewrite anything but its own retention columns (id=%).', OLD.id; + END IF; + IF NEW.revision <> OLD.revision + 1 OR NEW.last_modified_at < OLD.last_modified_at THEN + RAISE EXCEPTION 'agent_run_log_stream revision/time must advance monotonically (id=%, old_revision=%, new_revision=%).', OLD.id, OLD.revision, NEW.revision; + END IF; + RETURN NEW; + END IF; + + IF NEW.manifest_digest IS DISTINCT FROM OLD.manifest_digest + AND NOT (OLD.state = 'Open' AND NEW.state = 'Completed' AND NEW.schema_version = 3) THEN + RAISE EXCEPTION 'A log manifest receipt may only be set by v3 completion.'; + END IF; + IF NEW.schema_version = 3 AND NEW.state = 'Completed' THEN + IF NEW.content_digest IS NOT NULL OR NEW.content_digest_algorithm IS NOT NULL OR NEW.manifest_digest IS NULL + OR NOT EXISTS ( + SELECT 1 FROM agent_run_log_verification verification + WHERE verification.team_id = NEW.team_id AND verification.agent_run_id = NEW.agent_run_id + AND verification.stream_id = NEW.id AND verification.stream_revision = OLD.revision + AND verification.worker_fence_epoch = NEW.worker_fence_epoch AND verification.capture_session_id = NEW.capture_session_id + AND verification.segment_count = NEW.segment_count AND verification.total_bytes = NEW.total_bytes + AND verification.source_offset_bytes = NEW.source_offset_bytes AND verification.sealed_at IS NOT NULL + AND verification.next_segment_ordinal = NEW.segment_count + 1 AND verification.verified_bytes = NEW.total_bytes + AND verification.manifest_digest = NEW.manifest_digest) THEN + RAISE EXCEPTION 'v3 completion requires its exact sealed segment manifest; a whole SHA is not a manifest receipt.'; + END IF; + END IF; + + IF NEW.id IS DISTINCT FROM OLD.id OR NEW.team_id IS DISTINCT FROM OLD.team_id + OR NEW.agent_run_id IS DISTINCT FROM OLD.agent_run_id OR NEW.stream_kind IS DISTINCT FROM OLD.stream_kind + OR NEW.content_type IS DISTINCT FROM OLD.content_type OR NEW.content_encoding IS DISTINCT FROM OLD.content_encoding + OR NEW.capture_source IS DISTINCT FROM OLD.capture_source OR NEW.schema_version IS DISTINCT FROM OLD.schema_version + OR NEW.retention IS DISTINCT FROM OLD.retention OR NEW.expires_at IS DISTINCT FROM OLD.expires_at + OR NEW.created_at IS DISTINCT FROM OLD.created_at THEN + RAISE EXCEPTION 'agent_run_log_stream stable identity is immutable (id=%).', OLD.id; + END IF; + IF OLD.state <> 'Open' THEN + RAISE EXCEPTION 'agent_run_log_stream terminal state is immutable (id=%, state=%).', OLD.id, OLD.state; + END IF; + IF NEW.revision <> OLD.revision + 1 OR NEW.last_modified_at < OLD.last_modified_at THEN + RAISE EXCEPTION 'agent_run_log_stream revision/time must advance monotonically (id=%, old_revision=%, new_revision=%).', OLD.id, OLD.revision, NEW.revision; + END IF; + + IF NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id THEN + SELECT fence_epoch INTO current_fence FROM agent_run + WHERE team_id = NEW.team_id AND id = NEW.agent_run_id + FOR SHARE; + IF NOT FOUND OR current_fence <= 0 OR NEW.worker_fence_epoch IS DISTINCT FROM current_fence + OR NEW.worker_fence_epoch < COALESCE(OLD.worker_fence_epoch, 0) OR NEW.capture_session_id IS NULL THEN + RAISE EXCEPTION 'agent_run_log_stream stale or malformed capture claim rejected (run_id=%, current=%, attempted=%).', NEW.agent_run_id, current_fence, NEW.worker_fence_epoch; + END IF; + IF NEW.capture_session_id IS NOT DISTINCT FROM OLD.capture_session_id THEN + IF NEW.worker_fence_epoch <= COALESCE(OLD.worker_fence_epoch, 0) + OR NEW.capture_source_base_offset_bytes IS DISTINCT FROM OLD.capture_source_base_offset_bytes + OR NEW.capture_finalized_at IS DISTINCT FROM OLD.capture_finalized_at THEN + RAISE EXCEPTION 'agent_run_log_stream same-session reclaim requires a strictly newer fence and preserves source state (id=%).', OLD.id; + END IF; + ELSIF OLD.capture_finalized_at IS NULL OR NEW.capture_source_base_offset_bytes <> OLD.source_offset_bytes + OR NEW.capture_finalized_at IS NOT NULL THEN + RAISE EXCEPTION 'agent_run_log_stream stale or malformed capture claim rejected: next spool requires a finalized prior source and starts at its source head (id=%).', OLD.id; + END IF; + IF NEW.state IS DISTINCT FROM OLD.state OR NEW.segment_count IS DISTINCT FROM OLD.segment_count + OR NEW.total_bytes IS DISTINCT FROM OLD.total_bytes OR NEW.source_offset_bytes IS DISTINCT FROM OLD.source_offset_bytes + OR NEW.next_segment_ordinal IS DISTINCT FROM OLD.next_segment_ordinal + OR NEW.next_offset_bytes IS DISTINCT FROM OLD.next_offset_bytes OR NEW.completed_at IS DISTINCT FROM OLD.completed_at + OR NEW.error_code IS DISTINCT FROM OLD.error_code OR NEW.error_message IS DISTINCT FROM OLD.error_message + OR NEW.content_digest_algorithm IS DISTINCT FROM OLD.content_digest_algorithm + OR NEW.content_digest IS DISTINCT FROM OLD.content_digest THEN + RAISE EXCEPTION 'agent_run_log_stream capture claim cannot mutate byte or terminal state (id=%).', OLD.id; + END IF; + -- The two stall columns are deliberately ABSENT from that list: a claim is the one statement that may clear + -- them. A marker is one producer's statement about a segment it holds in memory, and a claim supersedes that + -- producer -- so a marker the claim inherited would have no one left to clear it and every reader would call + -- a healthily-capturing stream stalled forever. Do not add them here. + is_claim := TRUE; + END IF; + + IF NOT is_claim AND NEW.state = 'Open' AND OLD.capture_finalized_at IS NULL + AND NEW.capture_finalized_at IS NOT NULL THEN + IF NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id + OR NEW.segment_count IS DISTINCT FROM OLD.segment_count OR NEW.total_bytes IS DISTINCT FROM OLD.total_bytes + OR NEW.source_offset_bytes IS DISTINCT FROM OLD.source_offset_bytes + OR NEW.capture_source_base_offset_bytes IS DISTINCT FROM OLD.capture_source_base_offset_bytes + OR NEW.next_segment_ordinal IS DISTINCT FROM OLD.next_segment_ordinal + OR NEW.next_offset_bytes IS DISTINCT FROM OLD.next_offset_bytes OR NEW.completed_at IS NOT NULL + OR NEW.error_code IS NOT NULL OR NEW.error_message IS NOT NULL + OR NEW.content_digest_algorithm IS NOT NULL OR NEW.content_digest IS NOT NULL THEN + RAISE EXCEPTION 'agent_run_log_stream source finalization cannot rewrite its claim or byte head (id=%).', OLD.id; + END IF; + is_source_finalize := TRUE; + END IF; + + -- The FIFTH admissible update shape, and the reason this function is redefined: a REMOTE-STALL statement. The + -- four shapes above are a claim, a source finalization, an exact one-segment head advance, and a terminal + -- transition -- so a producer holding a segment behind a transient outage had no legal way to SAY so, and its + -- statement was refused as an append it never made. This arm admits exactly that statement and nothing more: the + -- two stall columns move, the revision advances as every update here must, and every claim, byte-head and + -- terminal column is required to be untouched. That is what keeps "my bytes are queued" from ever being + -- mistakable for "my bytes are stored". + IF NOT is_claim AND NOT is_source_finalize AND NEW.state = 'Open' + AND (NEW.remote_stall_since IS DISTINCT FROM OLD.remote_stall_since + OR NEW.remote_stall_code IS DISTINCT FROM OLD.remote_stall_code) THEN + IF NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id + OR NEW.segment_count <> OLD.segment_count OR NEW.total_bytes <> OLD.total_bytes + OR NEW.source_offset_bytes <> OLD.source_offset_bytes + OR NEW.capture_source_base_offset_bytes <> OLD.capture_source_base_offset_bytes + OR NEW.capture_finalized_at IS DISTINCT FROM OLD.capture_finalized_at + OR NEW.next_segment_ordinal <> OLD.next_segment_ordinal OR NEW.next_offset_bytes <> OLD.next_offset_bytes + OR NEW.completed_at IS NOT NULL OR NEW.error_code IS NOT NULL OR NEW.error_message IS NOT NULL + OR NEW.content_digest_algorithm IS NOT NULL OR NEW.content_digest IS NOT NULL + OR NEW.manifest_digest IS DISTINCT FROM OLD.manifest_digest THEN + RAISE EXCEPTION 'agent_run_log_stream remote-stall statement cannot rewrite its claim, byte head or terminal state (id=%).', OLD.id; + END IF; + is_remote_stall := TRUE; + END IF; + + IF NOT is_claim AND NOT is_source_finalize AND NOT is_remote_stall AND NEW.state = 'Open' THEN + IF NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id + OR OLD.capture_finalized_at IS NOT NULL OR NEW.capture_finalized_at IS NOT NULL + OR NEW.capture_source_base_offset_bytes IS DISTINCT FROM OLD.capture_source_base_offset_bytes + OR NEW.segment_count <> OLD.segment_count + 1 OR NEW.next_segment_ordinal <> OLD.next_segment_ordinal + 1 + OR NEW.total_bytes <= OLD.total_bytes OR NEW.next_offset_bytes <> NEW.total_bytes + OR NEW.source_offset_bytes <= OLD.source_offset_bytes + OR NEW.completed_at IS NOT NULL OR NEW.error_code IS NOT NULL OR NEW.error_message IS NOT NULL + OR NEW.content_digest_algorithm IS NOT NULL OR NEW.content_digest IS NOT NULL THEN + RAISE EXCEPTION 'agent_run_log_stream Open updates are exact one-segment head advances (id=%).', OLD.id; + END IF; + + SELECT * INTO appended FROM agent_run_log_segment + WHERE team_id = NEW.team_id AND stream_id = NEW.id AND agent_run_id = NEW.agent_run_id + AND segment_ordinal = OLD.next_segment_ordinal; + IF NOT FOUND OR appended.start_offset_bytes <> OLD.next_offset_bytes + OR appended.length_bytes <> NEW.total_bytes - OLD.total_bytes + OR appended.source_start_offset_bytes <> OLD.source_offset_bytes + OR appended.source_length_bytes <> NEW.source_offset_bytes - OLD.source_offset_bytes + OR appended.worker_fence_epoch IS DISTINCT FROM NEW.worker_fence_epoch + OR appended.capture_session_id IS DISTINCT FROM NEW.capture_session_id + OR appended.created_at > NEW.last_modified_at THEN + RAISE EXCEPTION 'agent_run_log_stream head advance requires its exact claimed append-only segment (id=%, ordinal=%).', OLD.id, OLD.next_segment_ordinal; + END IF; + ELSIF NOT is_claim AND NOT is_source_finalize AND NOT is_remote_stall THEN + SELECT fence_epoch INTO current_fence FROM agent_run + WHERE team_id = NEW.team_id AND id = NEW.agent_run_id + FOR SHARE; + IF NOT FOUND OR current_fence <= 0 THEN + RAISE EXCEPTION 'agent_run_log_stream terminal transition requires a live run fence (run_id=%, current=%).', NEW.agent_run_id, current_fence; + END IF; + -- The owner-loss exception, and the whole of it. A stream whose fence is STRICTLY behind the run's belongs to + -- a generation that is provably over, so no live capture session can still be appending to it and the only + -- thing left to say about it is that its owner is gone. CaptureFailed is the one state that says that; every + -- other terminal state is a claim about the bytes, which a generation that never read them cannot make. + is_owner_loss := OLD.worker_fence_epoch < current_fence AND NEW.state = 'CaptureFailed'; + IF NOT is_owner_loss AND OLD.worker_fence_epoch IS DISTINCT FROM current_fence THEN + RAISE EXCEPTION 'agent_run_log_stream terminal transition requires its current worker fence (run_id=%, current=%, claimed=%).', NEW.agent_run_id, current_fence, OLD.worker_fence_epoch; + END IF; + IF NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id + OR NEW.segment_count <> OLD.segment_count OR NEW.total_bytes <> OLD.total_bytes + OR NEW.source_offset_bytes <> OLD.source_offset_bytes + OR NEW.capture_source_base_offset_bytes <> OLD.capture_source_base_offset_bytes + OR NEW.capture_finalized_at IS DISTINCT FROM OLD.capture_finalized_at + OR NEW.next_segment_ordinal <> OLD.next_segment_ordinal OR NEW.next_offset_bytes <> OLD.next_offset_bytes + OR NEW.completed_at IS NULL OR NEW.completed_at < OLD.created_at THEN + RAISE EXCEPTION 'agent_run_log_stream terminal transition cannot rewrite its claim or byte head (id=%).', OLD.id; + END IF; + IF NEW.state = 'Completed' AND NEW.schema_version <> 3 AND (NEW.content_digest_algorithm IS DISTINCT FROM 'Sha256' + OR NEW.content_digest IS NULL OR octet_length(NEW.content_digest) IS DISTINCT FROM 32) THEN + RAISE EXCEPTION 'agent_run_log_stream Completed requires its verified SHA-256 content digest (id=%).', OLD.id; + END IF; + IF NEW.state = 'Completed' AND OLD.capture_finalized_at IS NULL THEN + RAISE EXCEPTION 'agent_run_log_stream Completed requires a durable final-drain receipt (id=%).', OLD.id; + END IF; + END IF; + + RETURN NEW; +END; +$$ LANGUAGE plpgsql; diff --git a/backend/src/CodeSpace.Core/Persistence/Entities/AgentRunLogStream.cs b/backend/src/CodeSpace.Core/Persistence/Entities/AgentRunLogStream.cs index b443feb9b..00150628e 100644 --- a/backend/src/CodeSpace.Core/Persistence/Entities/AgentRunLogStream.cs +++ b/backend/src/CodeSpace.Core/Persistence/Entities/AgentRunLogStream.cs @@ -43,6 +43,20 @@ public sealed class AgentRunLogStream : IEntity public DateTimeOffset? RemoteStallSince { get; set; } /// The typed refusal being waited out, in the capture bridge's error-code vocabulary. Never a second reason vocabulary, and never parsed. public string? RemoteStallCode { get; set; } + /// + /// The earliest instant this stream's bytes may be reclaimed, written by the retention reaper on the FIRST sweep + /// that found the stream terminal, past its rule and cited by nobody. Null means no sweep has ever proposed it — + /// which is what every stream a live worker is still capturing reads, and what the whole table read before the + /// plane existed. + /// + public DateTimeOffset? RetainUntil { get; set; } + + /// + /// When this stream's segment bytes were reclaimed. The head row deliberately outlives its bytes: a reader that + /// finds nothing cannot tell "purged by policy" from "lost", so the tombstone is what lets the Room say purged. + /// + public DateTimeOffset? PurgedAt { get; set; } + public ArtifactDigestAlgorithm? ContentDigestAlgorithm { get; set; } public byte[]? ContentDigest { get; set; } public byte[]? ManifestDigest { get; set; } diff --git a/backend/src/CodeSpace.Core/Persistence/Entities/PairedQualificationResultPin.cs b/backend/src/CodeSpace.Core/Persistence/Entities/PairedQualificationResultPin.cs new file mode 100644 index 000000000..257becae0 --- /dev/null +++ b/backend/src/CodeSpace.Core/Persistence/Entities/PairedQualificationResultPin.cs @@ -0,0 +1,46 @@ +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Core.Persistence.Entities; + +/// +/// One durable record a sealed cites, written in the seal's own transaction. +/// The retention planes read these rows as the proof that a record is still referenced — a result is a statement ABOUT +/// a set of runs, and nothing else in the schema expresses that. +/// +/// is a separate column rather than a typed use of because +/// ArtifactReferenceOracle enumerates reference sites by COLUMN NAME: an artifact pin has to sit in a column +/// ending in artifact_id, and nothing else may. Exactly one of the two is set, and the database's CHECK ties +/// which one to . +/// +public sealed class PairedQualificationResultPin : IEntity +{ + public Guid Id { get; set; } + public Guid ResultId { get; set; } + public DurablePinKind Kind { get; set; } + public Guid? PinnedId { get; set; } + public Guid? PinnedArtifactId { get; set; } + public DateTimeOffset PinnedAt { get; set; } + + /// The pinned record's id, whichever column carries it. + public Guid Target => PinnedId ?? PinnedArtifactId ?? Guid.Empty; + + public static PairedQualificationResultPin For(Guid resultId, DurablePinKind kind, Guid target, DateTimeOffset at) => new() + { + Id = Guid.NewGuid(), ResultId = resultId, Kind = kind, PinnedAt = at, + PinnedId = kind == DurablePinKind.Artifact ? null : target, + PinnedArtifactId = kind == DurablePinKind.Artifact ? target : null, + }; +} + +/// +/// What kind of record a pin names. Deliberately NOT : that enum is the set of classes +/// that have a retention RULE, and these are the kinds a qualification result can cite — an agent run has no rule at +/// all (it is identity, never collected) and is pinned anyway, because the other three are reached through it. +/// +public enum DurablePinKind +{ + LogStream = 1, + Artifact = 2, + CleanupReceipt = 3, + AgentRun = 4, +} diff --git a/backend/src/CodeSpace.Core/Persistence/EntityConfigurations/PairedQualificationResultPinConfiguration.cs b/backend/src/CodeSpace.Core/Persistence/EntityConfigurations/PairedQualificationResultPinConfiguration.cs new file mode 100644 index 000000000..6debc51a6 --- /dev/null +++ b/backend/src/CodeSpace.Core/Persistence/EntityConfigurations/PairedQualificationResultPinConfiguration.cs @@ -0,0 +1,27 @@ +using CodeSpace.Core.Persistence.Entities; +using Microsoft.EntityFrameworkCore; +using Microsoft.EntityFrameworkCore.Metadata.Builders; + +namespace CodeSpace.Core.Persistence.EntityConfigurations; + +/// Column names come from the global UseSnakeCaseNamingConvention(); the two CHECKs mirror migration 0235 exactly, and the unique index is an expression there (over both nullable target columns) so it is declared in SQL only. +public sealed class PairedQualificationResultPinConfiguration : IEntityTypeConfiguration +{ + public void Configure(EntityTypeBuilder builder) + { + builder.ToTable("paired_qualification_result_pin", table => + { + table.HasCheckConstraint("ck_paired_qualification_result_pin_kind", "kind IN ('LogStream', 'Artifact', 'CleanupReceipt', 'AgentRun')"); + table.HasCheckConstraint("ck_paired_qualification_result_pin_target", "num_nonnulls(pinned_id, pinned_artifact_id) = 1 AND (kind = 'Artifact') = (pinned_artifact_id IS NOT NULL)"); + }); + builder.HasKey(pin => pin.Id); + builder.Property(pin => pin.Id).ValueGeneratedNever(); + builder.Property(pin => pin.Kind).HasConversion().HasMaxLength(32); + builder.Ignore(pin => pin.Target); + + builder.HasOne().WithMany() + .HasForeignKey(pin => pin.ResultId) + .HasPrincipalKey(result => result.ObservationGroupId) + .OnDelete(DeleteBehavior.Cascade); + } +} diff --git a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs index b2af81bef..64d333886 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs @@ -58,10 +58,40 @@ public async Task SealAsync(PairedQualificationSealR ExpectedObservationCount = ExpectedCount(protocol, request.Manifest), ObservationCount = observations.Count, QualifiedForCapabilityClaim = outcome.QualifiedForCapabilityClaim, OutcomeJson = outcomeJson, }); + await PinCitedRecordsAsync(protocol.ObservationGroupId, observations, cancellationToken).ConfigureAwait(false); await _db.SaveChangesAsync(cancellationToken).ConfigureAwait(false); return sealedOutcome with { ResultDigest = resultDigest }; } + /// + /// Records which durable records this result cites, staged into the SAME SaveChanges as the result row. + /// That is the whole guarantee: a seal that commits has its pins, and a seal that does not leaves none — there is + /// no window in which a sealed result exists whose evidence a reaper is free to reclaim. + /// + /// The closure is column-derived, never inferred: the observations name their agent runs, and a run reaches + /// its log streams, its cleanup receipts and the offloaded payloads of its events. Migration 0235 backfills the + /// same four closures for results sealed before this existed. + /// + private async Task PinCitedRecordsAsync(Guid resultId, IReadOnlyList observations, CancellationToken cancellationToken) + { + var runIds = observations.Where(row => row.AgentRunId.HasValue).Select(row => row.AgentRunId!.Value).Distinct().ToList(); + + if (runIds.Count == 0) return; + + var now = DateTimeOffset.UtcNow; + var streamIds = await _db.AgentRunLogStream.AsNoTracking().Where(stream => runIds.Contains(stream.AgentRunId)).Select(stream => stream.Id).ToListAsync(cancellationToken).ConfigureAwait(false); + var receiptIds = await _db.AgentRunCleanupReceipt.AsNoTracking().Where(receipt => runIds.Contains(receipt.AgentRunId)).Select(receipt => receipt.Id).ToListAsync(cancellationToken).ConfigureAwait(false); + var artifactIds = await _db.AgentRunEvent.AsNoTracking().Where(row => runIds.Contains(row.AgentRunId) && row.DataArtifactId != null).Select(row => row.DataArtifactId!.Value).Distinct().ToListAsync(cancellationToken).ConfigureAwait(false); + + Pin(DurablePinKind.AgentRun, runIds); + Pin(DurablePinKind.LogStream, streamIds); + Pin(DurablePinKind.CleanupReceipt, receiptIds); + Pin(DurablePinKind.Artifact, artifactIds); + + void Pin(DurablePinKind kind, IEnumerable targets) => + _db.PairedQualificationResultPin.AddRange(targets.Select(target => PairedQualificationResultPin.For(resultId, kind, target, now))); + } + /// Verify the frozen runtime before the terminal row commits, so a seal minted on a substituted runtime leaves no result at all. Exempt for a replay — see . private async Task EnsureRuntimeUnchangedAsync(PairedQualificationSealRequest request, CancellationToken cancellationToken) { diff --git a/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs b/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs index 67dd4814a..a94c9dbee 100644 --- a/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs +++ b/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs @@ -576,8 +576,9 @@ private static string SpendDetail(decimal? usd, int unknown, string noun) RoomAgentLogStatus.Incomplete => 0, RoomAgentLogStatus.Stalled => 1, RoomAgentLogStatus.Finalizing => 2, - RoomAgentLogStatus.Captured => 3, - _ => 4, + RoomAgentLogStatus.Purged => 3, + RoomAgentLogStatus.Captured => 4, + _ => 5, }; private static string LogStatusWord(RoomAgentLogStatus status) => status switch @@ -585,6 +586,7 @@ private static string SpendDetail(decimal? usd, int unknown, string noun) RoomAgentLogStatus.Incomplete => "incomplete", RoomAgentLogStatus.Stalled => "held; storage unavailable", RoomAgentLogStatus.Finalizing => "finalizing", + RoomAgentLogStatus.Purged => "purged; retention window elapsed", RoomAgentLogStatus.Captured => "captured; integrity proof unavailable", _ => "integrity verified", }; @@ -596,6 +598,9 @@ private static string SpendDetail(decimal? usd, int unknown, string noun) // vocabulary for it would change every renderer for a state that is not a failure. RoomAgentLogStatus.Stalled => NarrativeTone.Info, RoomAgentLogStatus.Finalizing => NarrativeTone.Info, + // Info, not Error: nothing failed. The capture settled and the bytes were reclaimed on schedule; the WORD is + // what tells the reader they are gone. + RoomAgentLogStatus.Purged => NarrativeTone.Info, RoomAgentLogStatus.Captured => NarrativeTone.Info, _ => NarrativeTone.Success, }; diff --git a/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomProjector.cs b/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomProjector.cs index 75f4c2513..723574d32 100644 --- a/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomProjector.cs +++ b/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomProjector.cs @@ -218,7 +218,7 @@ private async Task> TerminalEvidence var logRows = await (from stream in _db.AgentRunLogStream.AsNoTracking() join agent in _db.AgentRun.AsNoTracking() on new { stream.TeamId, AgentRunId = stream.AgentRunId } equals new { agent.TeamId, AgentRunId = agent.Id } where stream.TeamId == teamId && agent.WorkflowRunId.HasValue && runIds.Contains(agent.WorkflowRunId.Value) - select new RunAgentLogRow(agent.WorkflowRunId.GetValueOrDefault(), stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null)) + select new RunAgentLogRow(agent.WorkflowRunId.GetValueOrDefault(), stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null, stream.PurgedAt != null)) .ToListAsync(cancellationToken).ConfigureAwait(false); var reservations = await _db.BudgetReservation.AsNoTracking() .Where(row => row.TeamId == teamId && runIds.Contains(row.WorkflowRunId)) @@ -743,7 +743,7 @@ private async Task> AgentLogsAsyn var rows = await _db.AgentRunLogStream.AsNoTracking() .Where(stream => stream.TeamId == teamId && agentIds.Contains(stream.AgentRunId)) - .Select(stream => new AgentLogRow(stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null)) + .Select(stream => new AgentLogRow(stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null, stream.PurgedAt != null)) .ToListAsync(cancellationToken).ConfigureAwait(false); return rows.GroupBy(row => row.AgentRunId).ToDictionary(group => group.Key, group => SummarizeLogs(group.ToList())); @@ -757,12 +757,19 @@ internal static RoomAgentLogSummary SummarizeLogs(IReadOnlyList row // bytes are queued in the sandbox spool. Saying "finalizing" through a storage incident is the reading an // operator acts on wrongly — it claims progress that is not happening and hides the one fact worth knowing. var stalled = rows.Count(row => row.State == AgentRunLogStreamState.Open && row.RemoteStalled); - var verified = rows.Count(row => row.State == AgentRunLogStreamState.Completed && row.SchemaVersion == 3 && row.HasManifestDigest); - var captured = rows.Count(row => row.State == AgentRunLogStreamState.Completed) - verified; - var status = incomplete.Count > 0 ? RoomAgentLogStatus.Incomplete : stalled > 0 ? RoomAgentLogStatus.Stalled : open > 0 ? RoomAgentLogStatus.Finalizing : captured > 0 ? RoomAgentLogStatus.Captured : RoomAgentLogStatus.Verified; + // A stream the retention plane reclaimed is counted apart from — never inside — captured or verified. Its + // manifest receipt is still on the row, so folding it as "integrity verified" would answer the operator's + // actual question ("can I read this?") with a proof about bytes that are gone. + var purged = rows.Count(row => row.State == AgentRunLogStreamState.Completed && row.Purged); + var settled = rows.Where(row => row.State == AgentRunLogStreamState.Completed && !row.Purged).ToList(); + var verified = settled.Count(row => row.SchemaVersion == 3 && row.HasManifestDigest); + var captured = settled.Count - verified; + var status = incomplete.Count > 0 ? RoomAgentLogStatus.Incomplete : stalled > 0 ? RoomAgentLogStatus.Stalled : open > 0 ? RoomAgentLogStatus.Finalizing + : purged > 0 ? RoomAgentLogStatus.Purged : captured > 0 ? RoomAgentLogStatus.Captured : RoomAgentLogStatus.Verified; var details = incomplete.GroupBy(row => row.State).OrderBy(group => LogStateRank(group.Key)).Select(group => $"{group.Count()} {LogStateWord(group.Key)}").ToList(); if (stalled > 0) details.Add($"{stalled} held; storage unavailable"); if (open - stalled > 0) details.Add($"{open - stalled} finalizing"); + if (purged > 0) details.Add($"{purged} purged; retention window elapsed"); if (captured > 0) details.Add($"{captured} captured; integrity proof unavailable"); if (verified > 0) details.Add($"{verified} integrity verified"); @@ -839,7 +846,7 @@ private static string DescribeRecovery(int orphaned, IReadOnlyList hosts private static readonly IReadOnlyDictionary EmptyTerminalEvidence = new Dictionary(); private static readonly IReadOnlyList EmptyAgentRows = Array.Empty(); - internal readonly record struct AgentLogRow(Guid AgentRunId, AgentRunLogStreamState State, int SchemaVersion, bool HasManifestDigest, bool RemoteStalled = false); + internal readonly record struct AgentLogRow(Guid AgentRunId, AgentRunLogStreamState State, int SchemaVersion, bool HasManifestDigest, bool RemoteStalled = false, bool Purged = false); /// One agent run of the turn, in the three columns a produced artifact has to be able to speak for. Internal so the producer fold is unit-pinned directly, not only through a full projection. internal readonly record struct AgentProducerRow(Guid AgentRunId, Messages.Enums.AgentRunStatus Status, string? ConfinementJson); @@ -847,9 +854,9 @@ private static string DescribeRecovery(int orphaned, IReadOnlyList hosts /// The per-UNIT facts every delivered artifact attaches — each unit's graded result, its log fold, and its producer record. Gathered once and handed to BOTH the deliveries and the deliverables projection, so a repository and a file can never attribute the same agent differently. private sealed record UnitTruth(IReadOnlyList Results, IReadOnlyDictionary Logs, IReadOnlyDictionary Producers); - private readonly record struct RunAgentLogRow(Guid RunId, Guid AgentRunId, AgentRunLogStreamState State, int SchemaVersion, bool HasManifestDigest, bool RemoteStalled) + private readonly record struct RunAgentLogRow(Guid RunId, Guid AgentRunId, AgentRunLogStreamState State, int SchemaVersion, bool HasManifestDigest, bool RemoteStalled, bool Purged) { - public AgentLogRow Log => new(AgentRunId, State, SchemaVersion, HasManifestDigest, RemoteStalled); + public AgentLogRow Log => new(AgentRunId, State, SchemaVersion, HasManifestDigest, RemoteStalled, Purged); } /// One agent row of a COLLAPSED turn, carrying its owning run so the batched read can be split per turn. Its is the identical row a fresh projection folds. @@ -1319,8 +1326,13 @@ private static RoomOracleProtection ProtectionOf(string? detail) return RoomOracleProtection.None; } - /// A stream that settled (Captured or Verified) reads complete; still Finalizing or Incomplete does not. - private static bool LogsAreComplete(RoomAgentLogStatus status) => status is RoomAgentLogStatus.Verified or RoomAgentLogStatus.Captured; + /// + /// A stream that settled (Captured, Verified, or settled and since Purged) reads complete; still Finalizing or + /// Incomplete does not. Purged belongs here because this field answers whether the capture SETTLED, not whether + /// the bytes are still on a destination — reading a reclaimed archive as an incomplete capture would retro-brand + /// a clean run as a broken one every time a retention window elapsed. The status word carries the bytes. + /// + private static bool LogsAreComplete(RoomAgentLogStatus status) => status is RoomAgentLogStatus.Verified or RoomAgentLogStatus.Captured or RoomAgentLogStatus.Purged; private static string? ClipVerificationDetail(string? detail) => string.IsNullOrEmpty(detail) ? null : detail.Length <= MaxVerificationDetailChars ? detail : detail[..MaxVerificationDetailChars].TrimEnd() + "…"; diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs index 8010dfc08..13619d4da 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs @@ -34,6 +34,7 @@ public sealed class ArtifactReferenceOracle : IArtifactReferenceOracle, IScopedD ("workflow_run_tool_call_attempt", "result_artifact_id"), ("workflow_run_tool_call_attempt", "error_artifact_id"), ("workflow_run_sensitive_record_payload", "ciphertext_artifact_id"), + ("paired_qualification_result_pin", "pinned_artifact_id"), ]; private static readonly string ExistsSql = BuildExistsSql(); diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs new file mode 100644 index 000000000..26b417c3d --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs @@ -0,0 +1,67 @@ +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Core.Services.Workflows.Retention; + +/// What a sweep decided to do with one durable record. +public enum DurableRetentionAction +{ + /// First observation of "nothing cites this" — record the quarantine deadline, remove nothing. + Quarantine, + + /// Both waits have elapsed and nothing cites the record. The ONLY action that removes anything. + Collect, + + /// Something cites the record. Keep. + Referenced, + + /// The status cannot be established. Keep. + Indeterminate, + + /// A scheduled wait (an age floor or a quarantine window). Keep, and re-ask when it elapses. + Wait, +} + +/// Everything one decision is allowed to depend on. A record so the decision stays a two-argument pure function. +public sealed record DurableRetentionObservation(DateTimeOffset TerminalAt, DateTimeOffset? RetainUntil, DurableReferenceVerdict Verdict, DateTimeOffset Now); + +/// +/// The ONE place that decides whether a durable record may be reclaimed. Pure, so every safety property is a table of +/// inputs rather than a claim about a distributed system: nothing here reaches a database, a clock or a store. +/// +/// Four properties live in this function and nowhere else. An unregistered class keeps. A record younger than +/// its class's age floor keeps, whatever its citation status. Any verdict other than a definite "nothing cites this" +/// keeps. And a first "nothing cites this" observation only ever quarantines — collection needs the quarantine window +/// to have elapsed on top of the age floor, which is a second, independent wait. +/// +public sealed record DurableRetentionDecision(DurableRetentionAction Action, string? Code, DateTimeOffset? RetainUntil) +{ + /// + /// Decide, from (null when the running policy registers no rule for the class) and + /// . Every branch except the last returns a KEEP; reaching + /// requires passing all of them. + /// + public static DurableRetentionDecision Decide(DurableRetentionRule? rule, DurableRetentionObservation observation) + { + ArgumentNullException.ThrowIfNull(observation); + + if (rule is null) return Indeterminate("retention-class-unregistered"); + + var eligibleAt = observation.TerminalAt.Add(rule.MinimumAge); + + if (observation.Now < eligibleAt) return Wait("age-floor-open", eligibleAt); + + if (observation.Verdict == DurableReferenceVerdict.Referenced) return Referenced(); + + if (observation.Verdict != DurableReferenceVerdict.Unreferenced) return Indeterminate("reference-status-indeterminate"); + + if (observation.RetainUntil is not { } retainUntil) return Quarantine(observation.Now.Add(rule.QuarantineWindow)); + + return observation.Now >= retainUntil ? Collect(retainUntil) : Wait("quarantine-window-open", retainUntil); + } + + public static DurableRetentionDecision Quarantine(DateTimeOffset retainUntil) => new(DurableRetentionAction.Quarantine, null, retainUntil); + public static DurableRetentionDecision Collect(DateTimeOffset retainUntil) => new(DurableRetentionAction.Collect, null, retainUntil); + public static DurableRetentionDecision Referenced() => new(DurableRetentionAction.Referenced, null, null); + public static DurableRetentionDecision Indeterminate(string code) => new(DurableRetentionAction.Indeterminate, code, null); + public static DurableRetentionDecision Wait(string code, DateTimeOffset until) => new(DurableRetentionAction.Wait, code, until); +} diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs new file mode 100644 index 000000000..c7999af80 --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs @@ -0,0 +1,53 @@ +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Core.Services.Workflows.Retention; + +/// +/// The retention policy for the durable records that are not artifacts: which classes exist and how long their records +/// are kept. Values are committed here and changed by a pull request — there is no environment override, because a +/// mistyped retention window is unrecoverable data loss and a code review is the control that belongs in front of it. +/// +/// An unregistered class is NOT an error a cursor can shrug off: returns null and the decision +/// settles Indeterminate, which keeps the record forever. That is what makes removing a class from this table safe. +/// +public static class DurableRetentionPolicy +{ + private static readonly TimeSpan Quarantine = TimeSpan.FromHours(24); + + /// + /// Thirty days after the stream reached its terminal capture state. Deliberately far longer than any run and + /// longer than any Room a human is still reading: the bytes this reclaims are an archive nobody opened, and the + /// floor costs storage rather than evidence. + /// + public static readonly DurableRetentionRule LogStream = new(DurableRecordClass.LogStream, TimeSpan.FromDays(30), Quarantine); + + /// Thirty days after the run went terminal. An Orphaned receipt is NEVER collected — it names a resource nobody reclaimed, so it is kept until something compensates it. + public static readonly DurableRetentionRule CleanupReceipt = new(DurableRecordClass.CleanupReceipt, TimeSpan.FromDays(30), Quarantine); + + /// The same floor as , because a gap describes a span of exactly one stream: outliving it would leave a hole with nothing to be a hole in, and predeceasing it would make an incomplete capture read complete. + public static readonly DurableRetentionRule CaptureGap = new(DurableRecordClass.CaptureGap, TimeSpan.FromDays(30), Quarantine); + + /// Half a year, and only for evidence no sealed result pins. A qualification claim is re-examined long after it was made, so its evidence outlives every other class here by a wide margin. + public static readonly DurableRetentionRule QualificationEvidence = new(DurableRecordClass.QualificationEvidence, TimeSpan.FromDays(180), Quarantine); + + /// Ninety days after reconciliation. A live reservation has no rule at all — it is not in a terminal state and therefore never a candidate. + public static readonly DurableRetentionRule BudgetReservation = new(DurableRecordClass.BudgetReservation, TimeSpan.FromDays(90), Quarantine); + + /// Seven days after the transfer saga ended. The record itself is small; what this class exists to reclaim is the staging object its terminal row still names. + public static readonly DurableRetentionRule TransferIntent = new(DurableRecordClass.TransferIntent, TimeSpan.FromDays(7), Quarantine); + + /// The committed table, public so a test can pin every literal value in it. + public static readonly IReadOnlyDictionary Rules = + new Dictionary + { + [LogStream.Class] = LogStream, + [CleanupReceipt.Class] = CleanupReceipt, + [CaptureGap.Class] = CaptureGap, + [QualificationEvidence.Class] = QualificationEvidence, + [BudgetReservation.Class] = BudgetReservation, + [TransferIntent.Class] = TransferIntent, + }; + + /// The rule for , or null when this build registers none — which every consumer reads as keep. + public static DurableRetentionRule? For(DurableRecordClass value) => Rules.TryGetValue(value, out var rule) ? rule : null; +} diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs new file mode 100644 index 000000000..62d46271f --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs @@ -0,0 +1,138 @@ +using CodeSpace.Core.Persistence.Db; +using CodeSpace.Messages.Retention; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Logging; + +namespace CodeSpace.Core.Services.Workflows.Retention; + +/// +/// The generic claim / decide / act loop, shared by every retention plane that is not the artifact ledger. +/// +/// What is structural rather than checked. (1) The cutoff a cursor claims against is derived HERE from +/// the class's own age floor, so a cursor cannot widen its candidate set past the policy. (2) The decision is +/// and nothing else, so every plane inherits the same two waits. +/// (3) Only removes anything, and it is reachable only after both waits +/// elapsed with a definite "nothing cites this". (4) A candidate that throws is logged and KEPT — the sweep never +/// converts an unanswered question into a deletion. +/// +/// Why there is no lease. The artifact reaper owns a ledger row and fences it with an owner and an epoch. +/// These planes have no ledger, and inventing one would mean a column per plane. The fence instead comes from the two +/// writes a collection actually makes: the byte removal goes through the CAS location's own durable claim, and the +/// tombstone is a conditional update on the record's revision. Two workers sweeping the same record at the same time +/// therefore remove the bytes at most once, and exactly one of them writes the tombstone; the loser settles nothing +/// and the next sweep meets a row it never held. +/// +public sealed class DurableRetentionReaper : IDurableRetentionReaper +{ + /// Records claimed per cursor per sweep. The cadence is hourly and every rule is measured in days, so the ceiling changes only how promptly a backlog drains — never what is collected. + private const int BatchSize = 200; + + private static readonly TimeSpan CandidateTimeout = TimeSpan.FromMinutes(2); + + private readonly DbContextOptions _dbOptions; + private readonly IReadOnlyList _cursors; + private readonly ILogger _logger; + + public DurableRetentionReaper(DbContextOptions dbOptions, IEnumerable cursors, ILogger logger) + { + _dbOptions = dbOptions; + _cursors = cursors.OrderBy(cursor => cursor.Class).ToArray(); + _logger = logger; + } + + public async Task SweepAsync(CancellationToken cancellationToken) + { + var now = await DatabaseClockAsync(cancellationToken).ConfigureAwait(false); + var counts = new SweepCounts(); + + foreach (var cursor in _cursors) + await SweepCursorAsync(cursor, now, counts, cancellationToken).ConfigureAwait(false); + + return counts.Summary(); + } + + /// One plane's bounded pass. A class the running policy does not register claims nothing at all, which is the same "keep forever" the decision returns for it. + private async Task SweepCursorAsync(IDurableRetentionCursor cursor, DateTimeOffset now, SweepCounts counts, CancellationToken cancellationToken) + { + var rule = DurableRetentionPolicy.For(cursor.Class); + + if (rule is null) + { + _logger.LogWarning("Durable retention: no rule is registered for {RecordClass}, so its records are kept and never claimed", cursor.Class); + return; + } + + var candidates = await cursor.ClaimAsync(now, now.Subtract(rule.MinimumAge), BatchSize, cancellationToken).ConfigureAwait(false); + counts.Claimed += candidates.Count; + + foreach (var candidate in candidates) + counts.Record(await SweepCandidateAsync(cursor, rule, candidate, now, cancellationToken).ConfigureAwait(false)); + } + + /// One candidate, start to finish. Every exit that is not a completed settlement keeps the record. + private async Task SweepCandidateAsync(IDurableRetentionCursor cursor, DurableRetentionRule rule, DurableRetentionCandidate candidate, + DateTimeOffset now, CancellationToken cancellationToken) + { + using var operation = CancellationTokenSource.CreateLinkedTokenSource(cancellationToken); + operation.CancelAfter(CandidateTimeout); + + try + { + var verdict = await cursor.ClassifyAsync(candidate, operation.Token).ConfigureAwait(false); + var decision = DurableRetentionDecision.Decide(rule, new DurableRetentionObservation(candidate.TerminalAt, candidate.RetainUntil, verdict, now)); + + return await ApplyAsync(cursor, candidate, decision, operation.Token).ConfigureAwait(false); + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) { throw; } + catch (Exception ex) + { + _logger.LogWarning(ex, "Durable retention: {RecordClass} {RecordId} could not be evaluated; the record is kept", cursor.Class, candidate.Id); + + return DurableRetentionAction.Indeterminate; + } + } + + /// A settlement the cursor could not complete — a lost race, a refused removal, a drain that needs another sweep — is reported as the keep it is. + private static async Task ApplyAsync(IDurableRetentionCursor cursor, DurableRetentionCandidate candidate, + DurableRetentionDecision decision, CancellationToken cancellationToken) + { + if (decision.Action is not (DurableRetentionAction.Quarantine or DurableRetentionAction.Collect)) return decision.Action; + + return await cursor.SettleAsync(candidate, decision, cancellationToken).ConfigureAwait(false) + ? decision.Action + : DurableRetentionAction.Indeterminate; + } + + /// The database's clock, not the worker's: every deadline these decisions compare against was written by a database clock too. + private async Task DatabaseClockAsync(CancellationToken cancellationToken) + { + await using var db = new CodeSpaceDbContext(_dbOptions); + + return await db.Database.SqlQueryRaw("SELECT clock_timestamp() AS \"Value\"").SingleAsync(cancellationToken).ConfigureAwait(false); + } + + private sealed class SweepCounts + { + public int Claimed { get; set; } + private int Quarantined { get; set; } + private int Collected { get; set; } + private int Referenced { get; set; } + private int Indeterminate { get; set; } + private int Waiting { get; set; } + + public void Record(DurableRetentionAction action) + { + if (action == DurableRetentionAction.Quarantine) Quarantined++; + else if (action == DurableRetentionAction.Collect) Collected++; + else if (action == DurableRetentionAction.Referenced) Referenced++; + else if (action == DurableRetentionAction.Wait) Waiting++; + else Indeterminate++; + } + + public DurableRetentionSweepSummary Summary() => new() + { + Claimed = Claimed, Quarantined = Quarantined, Collected = Collected, + Referenced = Referenced, Indeterminate = Indeterminate, Waiting = Waiting, + }; + } +} diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs new file mode 100644 index 000000000..7fba6588a --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs @@ -0,0 +1,53 @@ +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Core.Services.Workflows.Retention; + +/// Whether anything still cites one durable record, as far as its cursor can establish it. +public enum DurableReferenceVerdict +{ + /// Something still cites the record — a qualification pin, or another holder of the same bytes. Never collect. + Referenced = 1, + + /// Nothing cites the record at ANY enumerated site. A necessary condition for collection, never a sufficient one. + Unreferenced = 2, + + /// The question could not be answered. Read as "keep" everywhere it is consumed. + Indeterminate = 3, +} + +/// +/// One record a sweep is considering, in the only terms the generic loop needs: who it is, when its retention clock +/// started, and what an earlier sweep already decided about it. Plane-neutral on purpose — the cursor keeps whatever +/// else it needs to settle the row. +/// +public sealed record DurableRetentionCandidate(Guid Id, Guid TeamId, long Revision, DateTimeOffset TerminalAt, DateTimeOffset? RetainUntil); + +/// +/// One plane's answers to the three questions the reaper asks: which records are candidates, does anything still cite +/// this one, and apply this decision. Deliberately narrow (Rule 7): no policy, no clock, no batching — those belong to +/// the loop, so every plane inherits the same waits rather than re-deriving them. +/// +/// is fail-closed by contract: any failure to reach a citation site answers +/// , never . +/// returns false when the record moved under the sweep, which costs nothing — the next +/// sweep meets a row this one never held. +/// +public interface IDurableRetentionCursor +{ + DurableRecordClass Class { get; } + + /// + /// Records in a terminal state that went terminal at or before (the class's age + /// floor, computed by the loop), are not already collected, and are not deferred past . + /// Oldest first, at most . + /// + /// The deferral half is what keeps one unreclaimable record from owning a batch slot forever: a sweep that + /// cannot finish pushes the record's own retention deadline forward, and this query then skips it until then. + /// + Task> ClaimAsync(DateTimeOffset now, DateTimeOffset terminalBefore, int limit, CancellationToken cancellationToken); + + Task ClassifyAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken); + + /// Applies . Only and write anything; every other action is a keep with nothing to record. + Task SettleAsync(DurableRetentionCandidate candidate, DurableRetentionDecision decision, CancellationToken cancellationToken); +} diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionReaper.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionReaper.cs new file mode 100644 index 000000000..0cbf41d5b --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionReaper.cs @@ -0,0 +1,15 @@ +using CodeSpace.Core.DependencyInjection; +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Core.Services.Workflows.Retention; + +/// +/// Runs one BOUNDED retention sweep over every registered cursor. Bounded means three separate things, all of them +/// load-bearing: each cursor claims at most BatchSize records per sweep; a record is settled by its own +/// conditional write rather than a transaction spanning the batch; and a row that fails is logged and skipped, so one +/// unreachable destination cannot stop the rest of the sweep. +/// +public interface IDurableRetentionReaper : IScopedDependency +{ + Task SweepAsync(CancellationToken cancellationToken); +} diff --git a/backend/src/CodeSpace.Messages/Commands/Workflows/ReapExpiredDurableRecordsCommand.cs b/backend/src/CodeSpace.Messages/Commands/Workflows/ReapExpiredDurableRecordsCommand.cs new file mode 100644 index 000000000..a4135bf4e --- /dev/null +++ b/backend/src/CodeSpace.Messages/Commands/Workflows/ReapExpiredDurableRecordsCommand.cs @@ -0,0 +1,24 @@ +using CodeSpace.Messages.Mediation; +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Messages.Commands.Workflows; + +/// +/// Run one bounded durable-retention sweep: for every registered plane, claim records past their class's age floor, +/// establish whether anything still cites each one, and reclaim only those proven uncited past both waits. +/// +/// NOT tenant-scoped — system-wide reclamation that runs without an actor context. Fired by the recurring +/// reaper job; also sendable ad-hoc from an admin path or a test. +/// +/// NOT transactional (): each record is settled by its own conditional +/// write, and a reclamation that fails leaves only that record for the next pass. One command transaction around the +/// whole tick would let a single unreachable destination undo every other record's work — and it would hold a +/// transaction open across provider I/O. +/// +public sealed record ReapExpiredDurableRecordsCommand : ICommand, INonTransactionalCommand; + +/// The sweep's per-bucket counts, surfaced for logging and for the recurring job's result. +public sealed record ReapExpiredDurableRecordsResponse +{ + public required DurableRetentionSweepSummary Summary { get; init; } +} diff --git a/backend/src/CodeSpace.Messages/Dtos/Sessions/Room/RoomAgentLogStatus.cs b/backend/src/CodeSpace.Messages/Dtos/Sessions/Room/RoomAgentLogStatus.cs index fc6f312c9..e192d86b7 100644 --- a/backend/src/CodeSpace.Messages/Dtos/Sessions/Room/RoomAgentLogStatus.cs +++ b/backend/src/CodeSpace.Messages/Dtos/Sessions/Room/RoomAgentLogStatus.cs @@ -30,4 +30,12 @@ public enum RoomAgentLogStatus /// today, and a value shifted underneath a stored ordinal is unrecoverable. /// Stalled, + + /// + /// Settled, and its bytes have since been reclaimed by the retention plane. Distinct from every other member + /// because nothing went wrong: the capture completed, the window elapsed, and the head row survives precisely so a + /// reader is told that rather than meeting a stream whose bytes will not load. Appended for the same reason + /// was — no existing member's ordinal may move. + /// + Purged, } diff --git a/backend/src/CodeSpace.Messages/Retention/DurableRetention.cs b/backend/src/CodeSpace.Messages/Retention/DurableRetention.cs new file mode 100644 index 000000000..04d5637a4 --- /dev/null +++ b/backend/src/CodeSpace.Messages/Retention/DurableRetention.cs @@ -0,0 +1,60 @@ +namespace CodeSpace.Messages.Retention; + +/// +/// A class of durable record that has a retention rule. Membership is the same test the artifact plane applies to +/// ArtifactRetentionClass: a class exists only when the COMPLETE set of places that can still cite one of its +/// records is enumerable — a column, or a pin written in the citing statement's own transaction. A record whose +/// citations can also reach a JSON payload has no class and is therefore never a reap candidate. +/// +public enum DurableRecordClass +{ + /// An agent_run_log_stream head and the routed CAS bytes its segments name. + LogStream = 1, + + /// An agent_run_cleanup_receipt row (migration 0229). + CleanupReceipt = 2, + + /// A workflow_run_capture_gap row (migration 0231). Kept exactly as long as the stream whose missing span it describes. + CaptureGap = 3, + + /// The paired-qualification evidence tables (migrations 0218–0225) behind a sealed result. + QualificationEvidence = 4, + + /// A budget_reservation row (migration 0104) that has already been reconciled. + BudgetReservation = 5, + + /// An artifact_transfer_intent row (migration 0226) that reached a terminal state still holding a staging key. + TransferIntent = 6, +} + +/// +/// One class's committed rule. is the age floor measured from the record's own terminal +/// instant: below it the record is not even considered, so a citation that is still in flight cannot be outrun. +/// is the second, independent wait measured from the first observation of +/// "nothing cites this" — collection needs BOTH to have elapsed. +/// +public sealed record DurableRetentionRule(DurableRecordClass Class, TimeSpan MinimumAge, TimeSpan QuarantineWindow); + +/// What one bounded sweep did. Every claimed record lands in exactly one of these buckets. +public sealed record DurableRetentionSweepSummary +{ + public required int Claimed { get; init; } + + /// First observation of "nothing cites this": the quarantine deadline was recorded and nothing was removed. + public required int Quarantined { get; init; } + + /// Both waits elapsed with no citation. The only bucket that removed anything. + public required int Collected { get; init; } + + /// Something still cites the record. Kept. + public required int Referenced { get; init; } + + /// The question could not be answered, or the removal could not be completed. Kept. + public required int Indeterminate { get; init; } + + /// A scheduled wait — an age floor or a quarantine window that has not elapsed. Kept, and re-asked next sweep. + public required int Waiting { get; init; } + + public static DurableRetentionSweepSummary Empty { get; } = + new() { Claimed = 0, Quarantined = 0, Collected = 0, Referenced = 0, Indeterminate = 0, Waiting = 0 }; +} diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs new file mode 100644 index 000000000..891dda8bf --- /dev/null +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs @@ -0,0 +1,547 @@ +using Autofac; +using CodeSpace.Core.Persistence.Db; +using CodeSpace.Core.Persistence.Entities; +using CodeSpace.Core.Services.Agents.AgentRunLogging; +using CodeSpace.Core.Services.Agents.Eval.Benchmark; +using CodeSpace.Core.Services.Sessions.Room; +using CodeSpace.Core.Services.Credentials; +using CodeSpace.Core.Services.Workflows.Artifacts.Credentials; +using CodeSpace.Core.Services.Workflows.Artifacts.Providers.Local; +using CodeSpace.Core.Services.Workflows.Artifacts.Retention; +using CodeSpace.Core.Services.Workflows.Artifacts.Runtime; +using CodeSpace.Core.Services.Workflows.Retention; +using CodeSpace.Core.Services.Workflows.Retention.Logs; +using CodeSpace.IntegrationTests.Infrastructure; +using CodeSpace.Messages.Agents.Benchmark; +using CodeSpace.Messages.Artifacts; +using CodeSpace.Messages.Dtos.Sessions.Room; +using CodeSpace.Messages.Enums; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Logging.Abstractions; +using Npgsql; +using Shouldly; + +namespace CodeSpace.IntegrationTests.Workflows; + +/// +/// The durable-retention plane against a real database and a real local-rwx destination, where the parts that cannot +/// be unit-tested actually live: the guard trigger that decides which retention statements PostgreSQL admits, the pin +/// rows a seal writes in its own transaction, and the order in which bytes and their tombstone commit. +/// +/// The positive control at the top is what proves the counter-examples below it are not passing vacuously: each +/// of those builds a stream that is ONE property away from collectable and asserts its bytes survive. +/// +[Collection(PostgresCollection.Name)] +[Trait("Category", "Integration")] +public sealed class DurableRetentionReaperFlowTests : IDisposable +{ + private readonly PostgresFixture _fixture; + private readonly List _roots = []; + + public DurableRetentionReaperFlowTests(PostgresFixture fixture) => _fixture = fixture; + + /// + /// The positive control, and the ordering claim with it: the bytes leave the destination FIRST and the head row's + /// tombstone commits after, so a crash between them leaves bytes gone and a row the next sweep finishes — never a + /// row that says purged about bytes still being paid for. + /// + [Fact] + public async Task A_terminal_unpinned_log_stream_past_its_rule_is_purged_blobs_first_then_rows_and_the_Room_reads_purged() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "the archived transcript of a run nobody opened again"); + await AgeTerminalAsync(stream, TimeSpan.FromDays(31)); + + var first = await SweepAsync(); + + first.Quarantined.ShouldBeGreaterThanOrEqualTo(1, "the first uncited observation only starts the quarantine clock"); + (await StreamAsync(stream)).PurgedAt.ShouldBeNull("the sweep that first noticed the stream must not touch its bytes"); + (await StreamAsync(stream)).RetainUntil.ShouldNotBeNull("the quarantine deadline has to be durable — an in-memory one is no wait at all"); + (await BytesReadableAsync(world, stream)).ShouldBeTrue(); + + await ElapseQuarantineAsync(stream); + var second = await SweepAsync(); + + second.Collected.ShouldBeGreaterThanOrEqualTo(1); + var purged = await StreamAsync(stream); + purged.PurgedAt.ShouldNotBeNull("the head row is the tombstone: it outlives its bytes so a reader is told they were reclaimed"); + purged.State.ShouldBe(AgentRunLogStreamState.Completed, "a purge is not a capture verdict — the state it settled in is untouched"); + purged.TotalBytes.ShouldBeGreaterThan(0, "the byte head still records what was captured; only the bytes themselves are gone"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(0, "every location behind the stream's segments must be drained before the tombstone commits"); + (await BytesReadableAsync(world, stream)).ShouldBeFalse("the destination no longer holds the object"); + RoomProjector.SummarizeLogs([Row(purged)]).Status.ShouldBe(RoomAgentLogStatus.Purged, + "the Room must say purged rather than claim an integrity proof over bytes that are gone"); + } + + /// + /// The pin-table-first property, from the other side: a stream a sealed qualification result cites is never + /// collected, however long both windows have been open. Mutation: drop the pin probe in + /// LogStreamRetentionCursor.ClassifyAsync and this test reds with the stream purged. + /// + [Fact] + public async Task A_pinned_log_stream_past_its_rule_survives() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "evidence a capability claim was computed from"); + await SealResultAsync(world); + await AgeTerminalAsync(stream, TimeSpan.FromDays(400)); + + await SweepAsync(); + + (await StreamAsync(stream)).RetainUntil.ShouldBeNull("a cited stream is never even quarantined — the verdict is keep, not a wait"); + + // The strongest form of the claim: a quarantine deadline that has ALREADY elapsed, so the only thing left + // between the reaper and the bytes is the pin. + await ElapseQuarantineAsync(stream); + await SweepAsync(); + + var kept = await StreamAsync(stream); + kept.PurgedAt.ShouldBeNull("a sealed result still cites this stream; no elapsed window outranks that"); + (await BytesReadableAsync(world, stream)).ShouldBeTrue(); + (await UnpurgedLocationsAsync(world, stream)).ShouldBeGreaterThan(0); + } + + /// + /// The atomicity claim, measured rather than asserted: xmin is the id of the transaction that inserted a + /// row, so a result and its pins sharing one xmin IS "written in the same transaction". Mutation: move the + /// pin write to its own SaveChanges (or its own scope) and the two ids differ — which is the crash window + /// in which a sealed result exists whose evidence the reaper is free to reclaim. + /// + [Fact] + public async Task A_sealed_result_pins_every_stream_and_artifact_it_cites_in_one_transaction() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "one archived stream"); + var (receiptId, artifactId) = await SeedCitedRecordsAsync(world); + + var groupId = await SealResultAsync(world); + + var pins = await PinsAsync(groupId); + pins.Where(pin => pin.Kind == DurablePinKind.AgentRun).Select(pin => pin.Target).ShouldBe([world.AgentRunId]); + pins.Where(pin => pin.Kind == DurablePinKind.LogStream).Select(pin => pin.Target).ShouldBe([stream]); + pins.Where(pin => pin.Kind == DurablePinKind.CleanupReceipt).Select(pin => pin.Target).ShouldBe([receiptId]); + pins.Where(pin => pin.Kind == DurablePinKind.Artifact).Select(pin => pin.Target).ShouldBe([artifactId]); + pins.ShouldAllBe(pin => (pin.Kind == DurablePinKind.Artifact) == (pin.PinnedArtifactId != null), + "an artifact pin has to sit in the column the artifact oracle probes by name, and nothing else may"); + + var transactions = await InsertingTransactionsAsync(groupId); + transactions.Count.ShouldBe(1, $"the result and its {pins.Count} pins must commit together; they were inserted by {transactions.Count} transactions"); + + ArtifactReferenceOracle.ReferenceSites.ShouldContain(("paired_qualification_result_pin", "pinned_artifact_id"), + "an artifact a sealed result pins has to be a probed reference site, or the ARTIFACT reaper collects what this plane is protecting"); + } + + /// + /// Bytes before rows, proven by the failure: with the destination gone the purge is refused, and the sweep must + /// leave the head row exactly as it found it. Mutation: stamp the tombstone before (or without) the byte removal + /// and this reds — the Room would then read "purged" about an archive still sitting at the destination, and no + /// later sweep would ever reclaim it, because a purged row is never claimed again. + /// + [Fact] + public async Task A_refused_byte_purge_leaves_the_tombstone_unwritten() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "bytes at a destination that stops answering"); + await AgeTerminalAsync(stream, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(stream); + + var sweep = await SweepAgainstARefusingDestinationAsync(); + + sweep.Collected.ShouldBe(0, "nothing may be reported collected while the destination refuses to give the bytes up"); + var kept = await StreamAsync(stream); + kept.PurgedAt.ShouldBeNull("the tombstone must never run ahead of the bytes"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBeGreaterThan(0, "a refused delete leaves the location exactly where it was"); + kept.RetainUntil.ShouldNotBeNull().ShouldBeGreaterThan(DateTimeOffset.UtcNow, + "the stream is deferred rather than retried every tick, so one unreachable destination cannot own the batch"); + } + + /// + /// Migration 0236's arm, at the only place that can pin it: the database. Everything the guard refuses here is a + /// statement that would let a retention write masquerade as something else — or let a purge claim a wait it never + /// served. + /// + [Theory] + [InlineData("purged_at = now()", "cannot be purged without the retain_until")] + [InlineData("retain_until = now(), state = 'Corrupt'", "cannot rewrite anything but its own retention columns")] + [InlineData("retain_until = now(), total_bytes = 0", "cannot rewrite anything but its own retention columns")] + [InlineData("retain_until = now(), completed_at = now()", "cannot rewrite anything but its own retention columns")] + public async Task The_guard_refuses_a_retention_statement_that_says_more_than_retention(string smuggled, string refusal) + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "a settled stream"); + using var scope = _fixture.BeginScope(); + var sql = $"UPDATE agent_run_log_stream SET {smuggled}, revision = revision + 1, last_modified_at = now() WHERE team_id = {{0}} AND id = {{1}}"; + + var raised = await Should.ThrowAsync(async () => + await scope.Resolve().Database.ExecuteSqlRawAsync(sql, [world.TeamId, stream])); + + PostgresErrorOf(raised).MessageText.ShouldContain(refusal); + var row = await StreamAsync(stream); + row.RetainUntil.ShouldBeNull("a refused statement leaves the row exactly as it was"); + row.PurgedAt.ShouldBeNull(); + } + + /// A live capture is never a retention candidate, and the guard is what makes that true rather than a query predicate somebody can forget. + [Fact] + public async Task The_guard_refuses_a_retention_statement_on_an_open_stream() + { + var world = await SeedWorldAsync(); + using var scope = _fixture.BeginScope(); + var logs = Logs(scope); + var session = Guid.NewGuid(); + var opened = (await logs.OpenAsync(Open(world, session), CancellationToken.None)).ShouldBeOfType(); + + var raised = await Should.ThrowAsync(async () => await scope.Resolve().Database.ExecuteSqlRawAsync( + "UPDATE agent_run_log_stream SET retain_until = now(), revision = revision + 1, last_modified_at = now() WHERE team_id = {0} AND id = {1}", + [world.TeamId, opened.Metadata.StreamId])); + + PostgresErrorOf(raised).MessageText.ShouldContain("rejected on a live stream"); + } + + /// A purge is final: nothing may put a stream back into a state where it claims to hold bytes it gave up. + [Fact] + public async Task The_guard_refuses_to_un_purge_a_purged_stream() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "a stream that will be reclaimed"); + await AgeTerminalAsync(stream, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(stream); + await SweepAsync(); + (await StreamAsync(stream)).PurgedAt.ShouldNotBeNull(); + + using var scope = _fixture.BeginScope(); + var raised = await Should.ThrowAsync(async () => await scope.Resolve().Database.ExecuteSqlRawAsync( + "UPDATE agent_run_log_stream SET purged_at = NULL, revision = revision + 1, last_modified_at = now() WHERE team_id = {0} AND id = {1}", + [world.TeamId, stream])); + + PostgresErrorOf(raised).MessageText.ShouldContain("purge is final"); + } + + // ── The world, and the durable stream inside it ──────────────────────────────────────────────────────────────── + + /// One real capture through the real service: an open, one segment at a real local-rwx destination, a finalized source and a v3 completion. + private async Task CaptureAsync(World world, string text) + { + using var scope = _fixture.BeginScope(); + var logs = Logs(scope); + var session = Guid.NewGuid(); + var bytes = System.Text.Encoding.UTF8.GetBytes(text); + var opened = (await logs.OpenAsync(Open(world, session), CancellationToken.None)).ShouldBeOfType(); + var appended = (await logs.AppendAsync(new AgentRunLogAppendRequest + { + TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = opened.Metadata.StreamId, WorkerFenceEpoch = Fence, + CaptureSessionId = session, ExpectedSegmentOrdinal = 1, ExpectedOffsetBytes = 0, ExpectedSourceOffsetBytes = 0, + SourceLengthBytes = bytes.Length, StorageProfileId = world.StorageProfileId, StorageProfileRevision = 1, + ActorId = world.ActorId, Bytes = bytes, + }, CancellationToken.None)).ShouldBeOfType(); + var finalized = (await logs.FinalizeSourceAsync(new AgentRunLogFinalizeSourceRequest + { + TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = opened.Metadata.StreamId, WorkerFenceEpoch = Fence, + CaptureSessionId = session, ExpectedRevision = appended.Metadata.Revision, ExpectedSourceOffsetBytes = appended.Metadata.SourceOffsetBytes, + }, CancellationToken.None)).ShouldBeOfType(); + (await logs.CompleteAsync(new AgentRunLogCompleteRequest + { + TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = opened.Metadata.StreamId, WorkerFenceEpoch = Fence, + CaptureSessionId = session, ExpectedRevision = finalized.Metadata.Revision, + }, CancellationToken.None)).ShouldBeOfType(); + + return opened.Metadata.StreamId; + } + + /// + /// Time travel for the age floor only. The trigger is suspended for this one statement because the guard admits + /// no rewrite of completed_at at all — which is exactly the property the theory above pins — and the floor + /// is measured in days that no test can wait out. + /// + private async Task AgeTerminalAsync(Guid streamId, TimeSpan age) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream DISABLE TRIGGER agent_run_log_stream_enforce_invariants"); + try + { + await db.Database.ExecuteSqlRawAsync("UPDATE agent_run_log_stream SET completed_at = completed_at - {0}::interval WHERE id = {1}", + [$"{age.TotalSeconds} seconds", streamId]); + } + finally + { + await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream ENABLE TRIGGER agent_run_log_stream_enforce_invariants"); + } + } + + /// Pulls the quarantine deadline the sweep itself wrote into the past — through the guard, since that is a statement it admits. + private async Task ElapseQuarantineAsync(Guid streamId) + { + using var scope = _fixture.BeginScope(); + var updated = await scope.Resolve().Database.ExecuteSqlRawAsync( + "UPDATE agent_run_log_stream SET retain_until = now() - interval '1 second', revision = revision + 1, last_modified_at = now() WHERE id = {0}", [streamId]); + + updated.ShouldBe(1); + } + + /// + /// The real reaper over the real cursor, with only the CAS purge verdict swapped for the one a destination that + /// will not give the bytes up produces. Everything that decides whether the tombstone is written is production + /// code; the refusal is the single injected fact. + /// + private async Task SweepAgainstARefusingDestinationAsync() + { + using var scope = _fixture.BeginScope(); + var options = scope.Resolve>(); + var cursor = new LogStreamRetentionCursor(options, new RefusingPurgeCoordinator(scope.Resolve()), NullLogger.Instance); + + return await new DurableRetentionReaper(options, [cursor], NullLogger.Instance).SweepAsync(CancellationToken.None); + } + + /// A destination that answers every delete with a refusal and no effect — the shape a revoked key or a read-only bucket produces. + private sealed class RefusingPurgeCoordinator : IArtifactCasPurgeCoordinator + { + private readonly IArtifactCasPurgeCoordinator _inner; + + public RefusingPurgeCoordinator(IArtifactCasPurgeCoordinator inner) { _inner = inner; } + + public Task PurgeAsync(ArtifactCasPurgeRequest request, CancellationToken cancellationToken) => + Task.FromResult(new ArtifactCasPurgeResult.Rejected(new ArtifactCasProblem(ArtifactCasProblemCode.ProviderUnavailable, true))); + + public Task ClaimAsync(ArtifactCasPurgeRequest request, CancellationToken cancellationToken) => _inner.ClaimAsync(request, cancellationToken); + public Task DeleteAsync(ArtifactCasPurgeClaim claim, CancellationToken cancellationToken) => _inner.DeleteAsync(claim, cancellationToken); + public Task ReleaseAsync(ArtifactCasPurgeClaim claim, ArtifactCasReleaseEvidence evidence, CancellationToken cancellationToken) => _inner.ReleaseAsync(claim, evidence, cancellationToken); + public Task AbandonAsync(ArtifactCasPurgeClaim claim, CancellationToken cancellationToken) => _inner.AbandonAsync(claim, cancellationToken); + } + + // ── What a sealed result cites ───────────────────────────────────────────────────────────────────────────────── + + /// A cleanup receipt and an offloaded event payload for the same run, so the seal has all four pin kinds to write. + private async Task<(Guid ReceiptId, Guid ArtifactId)> SeedCitedRecordsAsync(World world) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + var artifactId = Guid.NewGuid(); + var receiptId = Guid.NewGuid(); + db.WorkflowArtifact.Add(new WorkflowArtifact + { + Id = artifactId, TeamId = world.TeamId, ContentType = "application/json", SizeBytes = 2, + Sha256 = Convert.ToHexString(System.Security.Cryptography.SHA256.HashData([1, 2])).ToLowerInvariant(), InlineBytes = [1, 2], + }); + db.AgentRunCleanupReceipt.Add(new AgentRunCleanupReceiptRecord + { + Id = receiptId, TeamId = world.TeamId, AgentRunId = world.AgentRunId, FenceEpoch = Fence, + Kind = Messages.Agents.Recovery.RunResourceKind.Spool, Outcome = Messages.Agents.Recovery.RunResourceOutcome.Completed, + RecordedByHost = "retention-test", RecordedAt = DateTimeOffset.UtcNow, + }); + await db.SaveChangesAsync(); + db.AgentRunEvent.Add(new AgentRunEvent + { + Id = Guid.NewGuid(), AgentRunId = world.AgentRunId, Kind = Messages.Agents.AgentEventKind.Warning, + Text = "an offloaded payload", DataArtifactId = artifactId, + }); + await db.SaveChangesAsync(); + + return (receiptId, artifactId); + } + + /// + /// A real seal through the real store: a hand-seeded protocol whose exact observation census is present, sealed as + /// a replay so the runtime gate — which is a different slice's concern — stays out of this one. + /// + private async Task SealResultAsync(World world) + { + var groupId = Guid.NewGuid(); + var manifest = new EvalSuiteManifest { Version = "sha256/retention-suite:v1", Cells = [new CorpusCellRef { TaskId = "task-a", Mode = BenchmarkMode.HarnessCli }] }; + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + db.PairedQualificationProtocol.Add(new PairedQualificationProtocol + { + ObservationGroupId = groupId, TeamId = world.TeamId, SuiteDigest = "sha256:hidden", SuiteVersion = manifest.Version, + CodeRevision = new string('c', 40), ControlModelRowId = world.ControlModelRowId, CandidateModelRowId = world.CandidateModelRowId, + RequiresCellAdmission = false, RequiresResultDigest = false, StatisticsVersion = PairedQualificationOutcome.StatisticsVersion, + Criterion = "Quality", SessionsPerCell = 1, MinimumIndependentClusters = 1, MinimumStrata = 1, MinimumRequiredExecutionClusters = 1, + MinimumEvaluatorHealth = 1, MaxCostUsdPerLaunch = 3m, MinimumQualityLift = 0.05, OrderingSeed = "frozen-order", + ProtocolDigest = Convert.ToHexString(System.Security.Cryptography.SHA256.HashData(groupId.ToByteArray())), + }); + foreach (var arm in new[] { "control", "candidate" }) + db.BenchmarkResultRecord.Add(Observation(world, groupId, manifest.Version, arm)); + await db.SaveChangesAsync(); + + var outcome = Outcome(groupId, manifest.Version, db.PairedQualificationProtocol.Single(row => row.ObservationGroupId == groupId).ProtocolDigest); + var store = new PairedQualificationResultStore(db, scope.Resolve()); + await store.SealAsync(new PairedQualificationSealRequest + { + ObservationGroupId = groupId, Manifest = manifest, Outcome = outcome, Source = PairedQualificationSealSource.Replay, + }, CancellationToken.None); + + return groupId; + } + + private BenchmarkResultRecord Observation(World world, Guid groupId, string suiteVersion, string arm) => new() + { + Id = Guid.NewGuid(), TeamId = world.TeamId, SuiteVersion = suiteVersion, TaskId = "task-a", Mode = BenchmarkMode.HarnessCli.ToString(), + Harness = "claude-code", Model = arm, ModelCredentialModelId = arm == "control" ? world.ControlModelRowId : world.CandidateModelRowId, + ObservationGroupId = groupId, ObservationArm = arm, ObservationSession = 0, AgentRunId = world.AgentRunId, + ObservedModel = arm, OutcomeState = "Solved", Solved = true, RunStatus = AgentRunStatus.Succeeded.ToString(), + MaxCostUsd = 3m, CostUsd = 1m, CostIndeterminate = false, GitSha = new string('c', 40), + }; + + private static PairedQualificationOutcome Outcome(Guid groupId, string suiteVersion, string protocolDigest) => new() + { + ObservationGroupId = groupId, ProtocolDigest = protocolDigest, CodeRevision = new string('c', 40), SuiteDigest = "sha256:hidden", + SuiteVersion = suiteVersion, IndependentClusters = 1, PairedCells = 1, RequiredExecutionClusters = 1, + Control = Arm(), Candidate = Arm(), QualityDifference = 0.1, QualityDifferenceLower95 = 0.06, + RequiredExecutionComplete = true, QualifiedForCapabilityClaim = true, BlockingReasons = [], Strata = [], InfraFailures = [], + }; + + private static PairedArmQualificationSummary Arm() => new() + { + Solved = 1, BudgetAdmissibleSolved = 1, Total = 1, InfraUnknown = 0, CostKnownCells = 1, + CapabilityVerdictCells = 1, ObservedModelCells = 1, EvaluatorHealth = 1, ObservedModels = ["model"], + }; + + // ── Reads ────────────────────────────────────────────────────────────────────────────────────────────────────── + + private async Task SweepAsync() + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().SweepAsync(CancellationToken.None); + } + + private async Task StreamAsync(Guid streamId) + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().AgentRunLogStream.AsNoTracking().SingleAsync(row => row.Id == streamId); + } + + private async Task> PinsAsync(Guid resultId) + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().PairedQualificationResultPin.AsNoTracking() + .Where(pin => pin.ResultId == resultId).OrderBy(pin => pin.Kind).ToListAsync(); + } + + /// The ids of the transactions that inserted the result and its pins. One id is the whole atomicity claim. + private async Task> InsertingTransactionsAsync(Guid resultId) + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().Database.SqlQueryRaw( + "SELECT DISTINCT xmin::text AS \"Value\" FROM paired_qualification_result WHERE observation_group_id = {0} " + + "UNION SELECT DISTINCT xmin::text FROM paired_qualification_result_pin WHERE result_id = {0}", resultId).ToListAsync(); + } + + private async Task UnpurgedLocationsAsync(World world, Guid streamId) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + + return await db.ArtifactLocation.AsNoTracking() + .Where(location => location.TeamId == world.TeamId && location.State != ArtifactLocationState.Purged && location.State != ArtifactLocationState.Deleted) + .Join(db.AgentRunLogSegment.AsNoTracking().Where(segment => segment.StreamId == streamId), + location => location.ArtifactObjectId, segment => segment.ArtifactObjectId, (location, _) => location.Id) + .CountAsync(); + } + + /// Whether the stream's bytes still come back through the production read path — the only reading of "the bytes are there" that matters. + private async Task BytesReadableAsync(World world, Guid streamId) + { + using var scope = _fixture.BeginScope(); + var stream = await StreamAsync(streamId); + var read = await Logs(scope).ReadRangeAsync(new AgentRunLogRangeRequest(world.TeamId, streamId, 0, checked((int)stream.TotalBytes)), CancellationToken.None); + + return read is AgentRunLogRangeResult.Available; + } + + private static RoomProjector.AgentLogRow Row(AgentRunLogStream stream) => + new(stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null, stream.PurgedAt != null); + + private static AgentRunLogService Logs(ILifetimeScope scope) => + new(scope.Resolve>(), scope.Resolve(), TimeProvider.System); + + private static AgentRunLogOpenRequest Open(World world, Guid session) => new() + { + TeamId = world.TeamId, AgentRunId = world.AgentRunId, WorkerFenceEpoch = Fence, CaptureSessionId = session, + StreamKind = AgentRunLogKinds.StandardOutput, ContentType = AgentRunLogRepresentations.PlainTextContentType, + ContentEncoding = AgentRunLogRepresentations.Utf8ContentEncoding, CaptureSource = "retention-test/v1", + }; + + /// EF's execution strategy wraps the server error; the guard's own message stays decisive. + private static PostgresException PostgresErrorOf(Exception error) + { + while (error is not PostgresException && error.InnerException is { } inner) error = inner; + return error.ShouldBeOfType(); + } + + private const long Fence = 7; + + /// A team whose log storage route is Active, whose credential resolves, and whose local-rwx root is real — so a refused purge is only ever the one the test caused. + private async Task SeedWorldAsync() + { + var actorId = Guid.NewGuid(); + var teamId = Guid.NewGuid(); + var profileId = Guid.NewGuid(); + var credentialId = Guid.NewGuid(); + var now = DateTimeOffset.UtcNow; + var root = NewRoot(); + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + db.User.Add(new User { Id = actorId, Email = $"durable-retention-{actorId:N}@test.local", Name = "Durable Retention" }); + db.Team.Add(new Team { Id = teamId, Slug = $"durable-retention-{teamId:N}", Name = "Durable Retention", Kind = TeamKind.Workspace }); + db.TeamMembership.Add(new TeamMembership { Id = Guid.NewGuid(), TeamId = teamId, UserId = actorId, Role = TeamRole.Owner }); + var credential = new StorageCredential + { + Id = credentialId, TeamId = teamId, StableName = $"durable-retention-{credentialId:N}", + CurrentRevision = 1, State = StorageCredentialState.Active, CreatedDate = now, CreatedBy = actorId, + }; + credential.Revisions.Add(new StorageCredentialRevision + { + Id = Guid.NewGuid(), TeamId = teamId, StorageCredentialId = credentialId, Revision = 1, + ProviderTypeKey = LocalRwxArtifactStorageDriverFactory.TypeKey, EncryptedPayload = scope.Resolve().Encrypt("{}"), + SafeHint = "safe", EnvelopeFingerprint = $"sha256:{new string('b', 64)}", CreatedDate = now, CreatedBy = actorId, + }); + db.StorageCredential.Add(credential); + var profile = new StorageProfile + { + Id = profileId, TeamId = teamId, StableName = $"durable-retention-{profileId:N}", State = StorageProfileState.Active, + CurrentRevision = 1, CreatedDate = now, CreatedBy = actorId, LastModifiedDate = now, LastModifiedBy = actorId, + }; + profile.Revisions.Add(new StorageProfileRevision + { + Id = Guid.NewGuid(), TeamId = teamId, StorageProfileId = profileId, Revision = 1, + ProviderTypeKey = LocalRwxArtifactStorageDriverFactory.TypeKey, NonSecretConfigJson = $"{{\"rootPath\":\"{root.Replace("\\", "\\\\")}\"}}", + CredentialRef = $"db:{credentialId:D}:1", NamespaceFingerprint = $"sha256:{Convert.ToHexString(System.Security.Cryptography.SHA256.HashData(profileId.ToByteArray())).ToLowerInvariant()}", + CreatedDate = now, CreatedBy = actorId, + }); + db.StorageProfile.Add(profile); + await db.SaveChangesAsync(); + var runId = Guid.NewGuid(); + db.AgentRun.Add(new AgentRun + { + Id = runId, TeamId = teamId, Harness = "test-harness", Status = AgentRunStatus.Running, TaskJson = "{}", + FenceEpoch = Fence, CreatedDate = now, CreatedBy = actorId, LastModifiedDate = now, LastModifiedBy = actorId, + }); + await db.SaveChangesAsync(); + + return new World(teamId, actorId, profileId, runId, Guid.NewGuid(), Guid.NewGuid()); + } + + private string NewRoot() + { + var root = Path.Combine(Path.GetTempPath(), $"codespace-durable-retention-{Guid.NewGuid():N}"); + Directory.CreateDirectory(root); + _roots.Add(root); + return root; + } + + public void Dispose() + { + foreach (var root in _roots) + { + try { if (Directory.Exists(root)) Directory.Delete(root, recursive: true); } catch { /* best-effort */ } + } + } + + private sealed record World(Guid TeamId, Guid ActorId, Guid StorageProfileId, Guid AgentRunId, Guid ControlModelRowId, Guid CandidateModelRowId); +} diff --git a/backend/tests/CodeSpace.UnitTests/Architecture/RequestAuthorizationInventoryTests.cs b/backend/tests/CodeSpace.UnitTests/Architecture/RequestAuthorizationInventoryTests.cs index 6ab9a89d4..eda533c36 100644 --- a/backend/tests/CodeSpace.UnitTests/Architecture/RequestAuthorizationInventoryTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Architecture/RequestAuthorizationInventoryTests.cs @@ -64,6 +64,7 @@ public class RequestAuthorizationInventoryTests ["ReapAgentRunSpoolsCommand"] = "sweep", ["ReapAgentRunOrphansCommand"] = "settles the cleanup receipts addressed to THIS host, across every team, by tearing down or observing host-local resources agent runs left behind when they were abandoned from another worker; which kernel objects and directories a machine still holds is a question no team can ask on its own behalf, and it reclaims what a run already finished with rather than changing any run's outcome.", ["ReapUnreferencedArtifactsCommand"] = "collects artifacts that a producer declared for retention and that no reference site points at, in bounded lease/fence batches; it is dispatched only by the system recurring job, acts for no user, and can never reach an artifact no producer declared.", + ["ReapExpiredDurableRecordsCommand"] = "reclaims the bytes behind durable records whose retention rule elapsed and that nothing — no column, no qualification pin — still cites, across every team in bounded batches; how long an archive is kept is a deployment-wide policy no team states on its own behalf, and it removes only what both waits and a fail-closed citation probe already cleared.", ["ReconcileAgentRunLogCapturesCommand"] = "reconciles exact durable AgentRun log-capture health in bounded lease/fence batches; it is dispatched only by the system recurring job and never acts for a user or changes an AgentRun outcome.", ["ResumeAbandonedArtifactTransfersCommand"] = "finishes artifact transfers whose worker died mid-flight, across every team, in bounded fence/lease batches; a parked transfer is unreachable to the team that started it, so this is a question no team can ask on its own behalf, and it starts nothing — it only completes or closes what a caller already began.", ["ReconcileRunDataManifestsCommand"] = "un-states the expectations of terminal runs whose producers declared more records than any of them accounted for, across every team, in bounded batches; whether a run's record was ever established is a question no team can ask on its own behalf, and it removes a claim rather than making one — no run can come out of it reading more complete than it went in.", diff --git a/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomLogSummaryTests.cs b/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomLogSummaryTests.cs index 49eda2238..ad75b02bb 100644 --- a/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomLogSummaryTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomLogSummaryTests.cs @@ -123,6 +123,32 @@ public void A_terminal_stream_still_outranks_a_held_one() summary.Detail.ShouldBe("2 streams · 1 capture failed · 1 held; storage unavailable"); } - private static RoomProjector.AgentLogRow Row(AgentRunLogStreamState state, int schemaVersion = 3, bool hasManifestDigest = false, bool remoteStalled = false) => - new(AgentId, state, schemaVersion, hasManifestDigest, remoteStalled); + /// A stream the retention plane reclaimed reads as purged, never as the integrity proof its manifest receipt still carries. + [Fact] + public void A_purged_stream_is_not_folded_as_integrity_verified() + { + var summary = RoomProjector.SummarizeLogs([Row(AgentRunLogStreamState.Completed, schemaVersion: 3, hasManifestDigest: true, purged: true)]); + + summary.ShouldBe(new RoomAgentLogSummary(RoomAgentLogStatus.Purged, 1, "1 stream · 1 purged; retention window elapsed")); + } + + [Fact] + public void A_readable_stream_outranks_a_purged_one_but_a_finalizing_one_outranks_both() + { + var purgedAndVerified = RoomProjector.SummarizeLogs([ + Row(AgentRunLogStreamState.Completed, schemaVersion: 3, hasManifestDigest: true, purged: true), + Row(AgentRunLogStreamState.Completed, schemaVersion: 3, hasManifestDigest: true), + ]); + + purgedAndVerified.Status.ShouldBe(RoomAgentLogStatus.Purged, "a reader has to be told some of this agent's log is gone, even when the rest verifies"); + purgedAndVerified.Detail.ShouldBe("2 streams · 1 purged; retention window elapsed · 1 integrity verified"); + + RoomProjector.SummarizeLogs([ + Row(AgentRunLogStreamState.Completed, schemaVersion: 3, hasManifestDigest: true, purged: true), + Row(AgentRunLogStreamState.Open), + ]).Status.ShouldBe(RoomAgentLogStatus.Finalizing, "a capture still in flight is the more urgent fact"); + } + + private static RoomProjector.AgentLogRow Row(AgentRunLogStreamState state, int schemaVersion = 3, bool hasManifestDigest = false, bool remoteStalled = false, bool purged = false) => + new(AgentId, state, schemaVersion, hasManifestDigest, remoteStalled, purged); } diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs index 13d718d15..3a397dfbd 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs @@ -38,6 +38,7 @@ public void Every_soft_link_to_an_artifact_is_a_probed_site() ("workflow_run_tool_call_attempt", "result_artifact_id"), ("workflow_run_tool_call_attempt", "error_artifact_id"), ("workflow_run_sensitive_record_payload", "ciphertext_artifact_id"), + ("paired_qualification_result_pin", "pinned_artifact_id"), }, ignoreOrder: true); } diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs new file mode 100644 index 000000000..33a2dd854 --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs @@ -0,0 +1,117 @@ +using CodeSpace.Core.Services.Workflows.Retention; +using CodeSpace.Messages.Retention; +using Shouldly; + +namespace CodeSpace.UnitTests.Workflows.Retention; + +/// +/// The committed rule table and the one function that decides whether a durable record may be reclaimed. The windows +/// are asserted as literals on purpose: they are the only numbers in this plane whose reduction destroys data, so +/// shortening one has to be a deliberate edit here as well as in the policy. +/// +/// Every test below names the mutation it catches, because a retention decision that is wrong in the permissive +/// direction has no second chance — the bytes are gone. +/// +[Trait("Category", "Unit")] +public sealed class DurableRetentionPolicyTests +{ + private static readonly DateTimeOffset Now = new(2026, 9, 18, 12, 0, 0, TimeSpan.Zero); + private static readonly DurableRetentionRule Rule = DurableRetentionPolicy.LogStream; + + [Theory] + [InlineData(DurableRecordClass.LogStream, 30)] + [InlineData(DurableRecordClass.CleanupReceipt, 30)] + [InlineData(DurableRecordClass.CaptureGap, 30)] + [InlineData(DurableRecordClass.QualificationEvidence, 180)] + [InlineData(DurableRecordClass.BudgetReservation, 90)] + [InlineData(DurableRecordClass.TransferIntent, 7)] + public void The_committed_rule_table_is_pinned_to_its_literal_windows(DurableRecordClass value, int minimumAgeDays) + { + var rule = DurableRetentionPolicy.For(value).ShouldNotBeNull(); + + rule.MinimumAge.ShouldBe(TimeSpan.FromDays(minimumAgeDays)); + rule.QuarantineWindow.ShouldBe(TimeSpan.FromHours(24), "every class waits a second, independent day after the first uncited observation"); + } + + [Fact] + public void Every_declared_class_has_a_rule_and_the_table_declares_nothing_else() + { + DurableRetentionPolicy.Rules.Keys.Order().ShouldBe(Enum.GetValues().Order(), + customMessage: "a class with no rule is never claimed, so adding one to the enum without a rule silently disables its plane"); + } + + [Fact] + public void A_class_the_policy_does_not_register_has_no_rule_and_therefore_keeps() + { + // The reaper reads a null rule as "claim nothing"; the decision reads it as Indeterminate. Both mean keep, + // which is what makes REMOVING a class from the table a safe operation rather than a purge. + DurableRetentionPolicy.For((DurableRecordClass)9999).ShouldBeNull(); + Decide(null, Now.AddDays(-400), null, DurableReferenceVerdict.Unreferenced).Action.ShouldBe(DurableRetentionAction.Indeterminate); + } + + [Fact] + public void Pinned_row_is_never_collected() + { + // Mutation: drop the pin check in the cursor (so a pinned record classifies Unreferenced) and this arm is the + // only thing left between a sealed qualification result and the evidence it was computed from. + var decision = Decide(Rule, Now.AddDays(-400), Now.AddDays(-100), DurableReferenceVerdict.Referenced); + + decision.Action.ShouldBe(DurableRetentionAction.Referenced, "a citation outranks every elapsed window, including a quarantine that is long past"); + decision.RetainUntil.ShouldBeNull(); + } + + [Fact] + public void First_unreferenced_observation_only_quarantines() + { + // Mutation: collect on the first uncited observation. The record would then be removed by the very sweep that + // first looked at it, with no durable window in which an operator or a late writer could intervene. + var decision = Decide(Rule, Now.AddDays(-400), null, DurableReferenceVerdict.Unreferenced); + + decision.Action.ShouldBe(DurableRetentionAction.Quarantine); + decision.RetainUntil.ShouldBe(Now.Add(Rule.QuarantineWindow)); + } + + [Fact] + public void An_open_quarantine_window_waits_rather_than_collecting() + { + var decision = Decide(Rule, Now.AddDays(-400), Now.AddHours(1), DurableReferenceVerdict.Unreferenced); + + decision.Action.ShouldBe(DurableRetentionAction.Wait); + decision.Code.ShouldBe("quarantine-window-open"); + } + + [Fact] + public void Both_waits_elapsed_with_no_citation_is_the_only_path_to_collect() + { + var decision = Decide(Rule, Now.AddDays(-400), Now.AddSeconds(-1), DurableReferenceVerdict.Unreferenced); + + decision.Action.ShouldBe(DurableRetentionAction.Collect); + } + + [Theory] + [InlineData(DurableReferenceVerdict.Unreferenced)] + [InlineData(DurableReferenceVerdict.Referenced)] + [InlineData(DurableReferenceVerdict.Indeterminate)] + public void Young_row_keeps_whatever_its_verdict(DurableReferenceVerdict verdict) + { + // Mutation: remove the age floor. A record whose citing write is still in flight — a seal mid-transaction, a + // receipt about to be upserted — would be observed uncited and quarantined before its writer ever committed. + var decision = Decide(Rule, Now.AddDays(-1), Now.AddDays(-100), verdict); + + decision.Action.ShouldBeOneOf(DurableRetentionAction.Wait, DurableRetentionAction.Referenced); + decision.Action.ShouldNotBe(DurableRetentionAction.Collect); + } + + [Fact] + public void An_unanswered_citation_question_keeps_the_record() + { + // Mutation: treat Indeterminate as Unreferenced. "I could not tell" must never resolve to "delete". + var decision = Decide(Rule, Now.AddDays(-400), Now.AddDays(-100), DurableReferenceVerdict.Indeterminate); + + decision.Action.ShouldBe(DurableRetentionAction.Indeterminate); + decision.Code.ShouldBe("reference-status-indeterminate"); + } + + private static DurableRetentionDecision Decide(DurableRetentionRule? rule, DateTimeOffset terminalAt, DateTimeOffset? retainUntil, DurableReferenceVerdict verdict) => + DurableRetentionDecision.Decide(rule, new DurableRetentionObservation(terminalAt, retainUntil, verdict, Now)); +} From 1613d02f1d057bb961efeaea03f36d23a4490fe1 Mon Sep 17 00:00:00 2001 From: "Mars.P" Date: Sat, 19 Sep 2026 07:47:53 +0800 Subject: [PATCH 2/4] Move the log-stream retention cursor out of a gitignored folder MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `.gitignore:19` ignores every directory named `Logs/`, so the cursor — the only file in this change that actually reclaims anything — was silently skipped by `git add` and never appeared in `git status`. The build and the tests were green locally because the file was on disk; CI would have failed to compile, because the integration test references the class. `Cursors/` is also the name rule 18.3 asks for: the sub-folder is named for the variant axis, which here is which plane a cursor sweeps, not which plane the first one happens to be. --- .../Cursors/LogStreamRetentionCursor.cs | 225 ++++++++++++++++++ .../DurableRetentionReaperFlowTests.cs | 2 +- 2 files changed, 226 insertions(+), 1 deletion(-) create mode 100644 backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs new file mode 100644 index 000000000..e4bfe98da --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs @@ -0,0 +1,225 @@ +using CodeSpace.Core.DependencyInjection; +using CodeSpace.Core.Persistence.Db; +using CodeSpace.Core.Persistence.Entities; +using CodeSpace.Core.Services.Workflows.Artifacts.Runtime; +using CodeSpace.Messages.Artifacts; +using CodeSpace.Messages.Constants; +using CodeSpace.Messages.Retention; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Logging; + +namespace CodeSpace.Core.Services.Workflows.Retention.Cursors; + +/// +/// Reclaims the routed CAS bytes behind a terminal agent-run log stream that nothing cites any more. +/// +/// The bytes go, the head row stays. A purge removes the segment objects' bytes and then stamps +/// purged_at on the stream head; the head, its segments and its verification manifest survive. That is +/// deliberate: a reader that finds no stream at all cannot tell "reclaimed by policy" from "capture lost it", and the +/// entire point of a retention plane is that reclamation is legible. The Room reads the tombstone and says purged. +/// +/// Order, and what a crash leaves. The bytes go first and the tombstone commits last. A crash in between +/// leaves bytes gone and no tombstone — the next sweep re-asks, finds the locations already Purged, and finishes the +/// stamp; the drain is idempotent by construction. The opposite order would leave bytes that no surviving row +/// remembers how to reach, which is the one outcome nothing can repair. +/// +/// What is never collected. A stream a sealed qualification result pins. A stream whose object another +/// stream's segment — or a workflow_artifact row — also names, because the CAS is content-addressed and those +/// are ONE physical object. And anything at all when the question could not be asked: every failure answers +/// , never "unreferenced". +/// +public sealed class LogStreamRetentionCursor : IDurableRetentionCursor, IScopedDependency +{ + /// + /// Objects purged per stream per sweep. A 4 GiB capture is thousands of segments, and a sweep that tried to drain + /// one in a single pass would hold a worker for minutes on provider I/O. Partial drains cost nothing: the tombstone + /// is only stamped once every object is gone, and the next sweep continues from whatever is left. + /// + private const int MaxObjectsPerSweep = 64; + + private readonly DbContextOptions _dbOptions; + private readonly IArtifactCasPurgeCoordinator _purge; + private readonly ILogger _logger; + + public LogStreamRetentionCursor(DbContextOptions dbOptions, IArtifactCasPurgeCoordinator purge, ILogger logger) + { + _dbOptions = dbOptions; + _purge = purge; + _logger = logger; + } + + public DurableRecordClass Class => DurableRecordClass.LogStream; + + public async Task> ClaimAsync(DateTimeOffset now, DateTimeOffset terminalBefore, int limit, CancellationToken cancellationToken) + { + await using var db = CreateDb(); + + return await db.AgentRunLogStream.AsNoTracking() + .Where(stream => stream.State != AgentRunLogStreamState.Open && stream.PurgedAt == null + && stream.CompletedAt != null && stream.CompletedAt <= terminalBefore + && (stream.RetainUntil == null || stream.RetainUntil <= now)) + .OrderBy(stream => stream.CompletedAt).ThenBy(stream => stream.Id) + .Take(limit) + .Select(stream => new DurableRetentionCandidate(stream.Id, stream.TeamId, stream.Revision, stream.CompletedAt!.Value, stream.RetainUntil)) + .ToListAsync(cancellationToken).ConfigureAwait(false); + } + + /// Fail-closed: any failure to probe a citation site answers indeterminate, which every consumer reads as keep. + public async Task ClassifyAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken) + { + ArgumentNullException.ThrowIfNull(candidate); + + try + { + await using var db = CreateDb(); + + if (await IsPinnedAsync(db, candidate.Id, cancellationToken).ConfigureAwait(false)) return DurableReferenceVerdict.Referenced; + + return await SharesBytesAsync(db, candidate, cancellationToken).ConfigureAwait(false) + ? DurableReferenceVerdict.Referenced + : DurableReferenceVerdict.Unreferenced; + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) { throw; } + catch (Exception ex) + { + _logger.LogWarning(ex, "Log stream {StreamId}: citation sites could not be probed; the stream is kept", candidate.Id); + + return DurableReferenceVerdict.Indeterminate; + } + } + + public async Task SettleAsync(DurableRetentionCandidate candidate, DurableRetentionDecision decision, CancellationToken cancellationToken) + { + ArgumentNullException.ThrowIfNull(candidate); + ArgumentNullException.ThrowIfNull(decision); + + if (decision.Action == DurableRetentionAction.Quarantine) return await StampAsync(candidate, decision.RetainUntil, null, cancellationToken).ConfigureAwait(false); + if (decision.Action != DurableRetentionAction.Collect) return false; + + return await CollectAsync(candidate, decision, cancellationToken).ConfigureAwait(false); + } + + /// Bytes first, then the tombstone. A drain that did not finish defers the stream instead of stamping it, so the claim query stops returning it every hour. + private async Task CollectAsync(DurableRetentionCandidate candidate, DurableRetentionDecision decision, CancellationToken cancellationToken) + { + var drained = await PurgeBytesAsync(candidate, cancellationToken).ConfigureAwait(false); + var now = await DatabaseClockAsync(cancellationToken).ConfigureAwait(false); + + if (!drained) + { + await StampAsync(candidate, now.Add(DurableRetentionPolicy.LogStream.QuarantineWindow), null, cancellationToken).ConfigureAwait(false); + + return false; + } + + var stamped = await StampAsync(candidate, decision.RetainUntil, now, cancellationToken).ConfigureAwait(false); + + if (stamped) + _logger.LogInformation("Log stream {StreamId} purged for team {TeamId}: its segment bytes were reclaimed and the head row now reads purged", candidate.Id, candidate.TeamId); + + return stamped; + } + + /// + /// Removes every object this stream's segments still name, bounded per sweep. True only when NOTHING is left to + /// remove — the tombstone must not claim more than the destination actually gave up. + /// + private async Task PurgeBytesAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken) + { + var objectIds = await ReclaimableObjectsAsync(candidate, MaxObjectsPerSweep + 1, cancellationToken).ConfigureAwait(false); + + if (objectIds.Count == 0) return true; + + foreach (var objectId in objectIds.Take(MaxObjectsPerSweep)) + { + if (!await PurgeObjectAsync(candidate, objectId, cancellationToken).ConfigureAwait(false)) return false; + } + + return objectIds.Count <= MaxObjectsPerSweep; + } + + private async Task PurgeObjectAsync(DurableRetentionCandidate candidate, Guid objectId, CancellationToken cancellationToken) + { + var request = new ArtifactCasPurgeRequest { TeamId = candidate.TeamId, ArtifactObjectId = objectId, ActorId = SystemUsers.SeederId }; + var outcome = await _purge.PurgeAsync(request, cancellationToken).ConfigureAwait(false); + + if (outcome is ArtifactCasPurgeResult.Purged) return true; + + var rejected = (ArtifactCasPurgeResult.Rejected)outcome; + _logger.LogWarning("Log stream {StreamId}: object {ObjectId} was not reclaimed ('{Problem}'); the stream keeps its bytes and its head row", candidate.Id, objectId, rejected.Problem.Code); + + return false; + } + + /// + /// The objects whose bytes are still somewhere, read inside the collecting pass rather than carried from the + /// decision: a location the previous partial drain already purged is not asked about twice, which is what makes a + /// resumed drain converge. + /// + private async Task> ReclaimableObjectsAsync(DurableRetentionCandidate candidate, int limit, CancellationToken cancellationToken) + { + await using var db = CreateDb(); + + return await db.AgentRunLogSegment.AsNoTracking() + .Where(segment => segment.TeamId == candidate.TeamId && segment.StreamId == candidate.Id) + .Select(segment => segment.ArtifactObjectId) + .Distinct() + .Where(objectId => db.ArtifactLocation.Any(location => location.TeamId == candidate.TeamId && location.ArtifactObjectId == objectId + && location.State != ArtifactLocationState.Purged && location.State != ArtifactLocationState.Deleted)) + .OrderBy(objectId => objectId) + .Take(limit) + .ToListAsync(cancellationToken).ConfigureAwait(false); + } + + private static async Task IsPinnedAsync(CodeSpaceDbContext db, Guid streamId, CancellationToken cancellationToken) => + await db.PairedQualificationResultPin.AsNoTracking() + .AnyAsync(pin => pin.Kind == DurablePinKind.LogStream && pin.PinnedId == streamId, cancellationToken).ConfigureAwait(false); + + /// + /// Whether anything else holds the same physical bytes. The CAS addresses an object by content, so two streams + /// that captured identical bytes are ONE object with two names, and removing it for one takes the other's content + /// too. Keeping is always safe here; collecting is not. + /// + private static async Task SharesBytesAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, CancellationToken cancellationToken) + { + var objects = db.AgentRunLogSegment.AsNoTracking() + .Where(segment => segment.TeamId == candidate.TeamId && segment.StreamId == candidate.Id) + .Select(segment => segment.ArtifactObjectId); + + if (await db.AgentRunLogSegment.AsNoTracking().AnyAsync(segment => segment.StreamId != candidate.Id && objects.Contains(segment.ArtifactObjectId), cancellationToken).ConfigureAwait(false)) + return true; + + return await db.WorkflowArtifact.AsNoTracking() + .AnyAsync(artifact => artifact.CasArtifactObjectId != null && objects.Contains(artifact.CasArtifactObjectId.Value), cancellationToken).ConfigureAwait(false); + } + + /// + /// The one write this cursor makes, and the only shape migration 0236's guard admits: the two retention columns, + /// a revision that advances, nothing else. The revision is also the fence — a stream another worker moved since + /// the claim matches nothing, and that sweep simply settles nothing. + /// + private async Task StampAsync(DurableRetentionCandidate candidate, DateTimeOffset? retainUntil, DateTimeOffset? purgedAt, CancellationToken cancellationToken) + { + var now = purgedAt ?? await DatabaseClockAsync(cancellationToken).ConfigureAwait(false); + await using var db = CreateDb(); + + var updated = await db.AgentRunLogStream + .Where(stream => stream.Id == candidate.Id && stream.TeamId == candidate.TeamId && stream.Revision == candidate.Revision && stream.PurgedAt == null) + .ExecuteUpdateAsync(set => set + .SetProperty(stream => stream.RetainUntil, retainUntil) + .SetProperty(stream => stream.PurgedAt, purgedAt) + .SetProperty(stream => stream.Revision, stream => stream.Revision + 1) + .SetProperty(stream => stream.LastModifiedAt, now), cancellationToken).ConfigureAwait(false); + + return updated == 1; + } + + private CodeSpaceDbContext CreateDb() => new(_dbOptions); + + private async Task DatabaseClockAsync(CancellationToken cancellationToken) + { + await using var db = CreateDb(); + + return await db.Database.SqlQueryRaw("SELECT clock_timestamp() AS \"Value\"").SingleAsync(cancellationToken).ConfigureAwait(false); + } +} diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs index 891dda8bf..9f800edae 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs @@ -10,7 +10,7 @@ using CodeSpace.Core.Services.Workflows.Artifacts.Retention; using CodeSpace.Core.Services.Workflows.Artifacts.Runtime; using CodeSpace.Core.Services.Workflows.Retention; -using CodeSpace.Core.Services.Workflows.Retention.Logs; +using CodeSpace.Core.Services.Workflows.Retention.Cursors; using CodeSpace.IntegrationTests.Infrastructure; using CodeSpace.Messages.Agents.Benchmark; using CodeSpace.Messages.Artifacts; From a69c4d5a25f8d5da712b878668dfc3eaff38841f Mon Sep 17 00:00:00 2001 From: "Mars.P" Date: Sat, 19 Sep 2026 12:31:07 +0800 Subject: [PATCH 3/4] Keep the reaper honest about what it kept and what it took MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A review of the first cut found four ways the plane could look like it was working while doing nothing, or say something it had not earned. A keep wrote nothing at all, so a stream nothing can ever reclaim — one a sealed result pins, or one whose bytes another stream also names — was re-claimed by the oldest-first query on every tick. A few hundred of those deployment-wide would have filled the batch for ever while the sweep reported a healthy claim count. Every decision is now settled, including the keeps, and the claim query leaves a record alone for its recheck interval. The read path answered a policy reclamation in the vocabulary of data loss: the segment walk found no available location and reported ArtifactMissing, byte-identical to an object that vanished. Worse, the metadata kept projecting the surviving manifest receipt as an integrity proof over bytes that were gone. A purged stream now answers with its own code before the walk, and withholds the proof. The citer list was neither complete nor pinned. It is enumerated, covered by a drift detector over the EF model, and it deliberately does NOT include a transfer intent: ck_artifact_transfer_intent_outcome makes that column null on every non-terminal row, so the only rows naming an object are finished transfers — and a transfer is what wrote each segment, so probing it would have kept every log stream in the deployment for ever. Three smaller reversals of meaning: a stream whose bytes were lost by some other path is no longer tombstoned as purged; a drain that is still making progress no longer backs off like a refusal; and the guard now requires the quarantine to have been recorded by an EARLIER statement, so one sweep cannot both propose and execute a collection. The rule table lists one class, because one cursor exists. The five classes it advertised without one are gone until their cursors arrive. --- .../Controllers/AgentsController.cs | 4 +- .../Agents/AgentRunLogQueryHandlers.cs | 1 + .../0236_agent_run_log_stream_retention.sql | 57 +- .../AgentRunLogging/AgentRunLogService.cs | 13 +- .../AgentRunLogging/IAgentRunLogService.cs | 15 + .../PairedQualificationResultStore.cs | 13 +- .../Cursors/LogStreamRetentionCursor.cs | 271 +++++-- .../Retention/DurableRetentionDecision.cs | 24 +- .../Retention/DurableRetentionPolicy.cs | 40 +- .../Retention/DurableRetentionReaper.cs | 33 +- .../Retention/IDurableRetentionCursor.cs | 36 +- .../Dtos/Agents/AgentRunLogDtos.cs | 3 + .../Retention/DurableRetention.cs | 51 +- .../DurableRetentionReaperFlowTests.cs | 670 +++++++++++++++--- .../Persistence/AgentRunLogSchemaTests.cs | 5 + .../Retention/DurableRetentionPolicyTests.cs | 45 +- .../Retention/LogStreamCitationSitesTests.cs | 77 ++ frontend/src/api/sessions.ts | 2 +- .../components/sessions/SessionRoomView.tsx | 1 + 19 files changed, 1074 insertions(+), 287 deletions(-) create mode 100644 backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs diff --git a/backend/src/CodeSpace.Api/Controllers/AgentsController.cs b/backend/src/CodeSpace.Api/Controllers/AgentsController.cs index c1850dfff..f443b118a 100644 --- a/backend/src/CodeSpace.Api/Controllers/AgentsController.cs +++ b/backend/src/CodeSpace.Api/Controllers/AgentsController.cs @@ -146,7 +146,9 @@ public async Task ReadRunLog([FromRoute] Guid agentRunId, [FromRo AgentRunLogReadAvailability.InvalidRange => StatusCodes.Status400BadRequest, AgentRunLogReadAvailability.AccessDenied => StatusCodes.Status424FailedDependency, AgentRunLogReadAvailability.BackendUnavailable or AgentRunLogReadAvailability.ProviderTimeout => StatusCodes.Status503ServiceUnavailable, - AgentRunLogReadAvailability.PhysicalObjectMissing or AgentRunLogReadAvailability.IntegrityFailure => StatusCodes.Status410Gone, + // 410 for all three: the bytes are not coming back. The availability code beside it is what distinguishes a + // policy reclamation from a loss, which the status alone cannot say. + AgentRunLogReadAvailability.PhysicalObjectMissing or AgentRunLogReadAvailability.IntegrityFailure or AgentRunLogReadAvailability.Purged => StatusCodes.Status410Gone, _ => StatusCodes.Status422UnprocessableEntity, }; diff --git a/backend/src/CodeSpace.Core/Handlers/QueryHandlers/Agents/AgentRunLogQueryHandlers.cs b/backend/src/CodeSpace.Core/Handlers/QueryHandlers/Agents/AgentRunLogQueryHandlers.cs index 6be8ef105..e4ca9bd26 100644 --- a/backend/src/CodeSpace.Core/Handlers/QueryHandlers/Agents/AgentRunLogQueryHandlers.cs +++ b/backend/src/CodeSpace.Core/Handlers/QueryHandlers/Agents/AgentRunLogQueryHandlers.cs @@ -208,6 +208,7 @@ public static AgentRunLogRangeRead Available(AgentRunLogRangeResult.Available va { AgentRunLogProblemCode.InvalidRequest => AgentRunLogReadAvailability.InvalidRange, AgentRunLogProblemCode.Missing or AgentRunLogProblemCode.ArtifactMissing => AgentRunLogReadAvailability.PhysicalObjectMissing, + AgentRunLogProblemCode.Purged => AgentRunLogReadAvailability.Purged, AgentRunLogProblemCode.ArtifactCorrupt => AgentRunLogReadAvailability.IntegrityFailure, AgentRunLogProblemCode.AccessDenied => AgentRunLogReadAvailability.AccessDenied, AgentRunLogProblemCode.ProviderTimeout => AgentRunLogReadAvailability.ProviderTimeout, diff --git a/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql index 36c30d8a0..e66a796f7 100644 --- a/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql +++ b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql @@ -15,16 +15,23 @@ -- This migration adds ONE admissible shape, the sixth, and makes it as narrow as the row allows: a RETENTION -- STATEMENT may move retain_until and purged_at on a terminal stream and NOTHING else. It cannot change the state, the -- claim, the byte head, the source offsets, the finalization receipt, the digests or the error — so a purge can never --- be mistaken for a capture verdict, and the durable prefix's metadata stays exactly as readable as it was. Three --- further rules ride with it, each of which exists because its absence is silent data loss: +-- be mistaken for a capture verdict, and the durable prefix's metadata stays exactly as readable as it was. A +-- statement that touches anything else still reads the refusal it always did, word for word, because the arm +-- reproduces it rather than replacing it. Three further rules ride with the arm, each of which exists because its +-- absence is silent data loss: -- -- * A stream that is still Open is never a retention candidate. Its bytes belong to a live capture session, so the -- columns are refused there rather than left to fall through an arm that does not enumerate them. --- * A purge requires a retain_until that was already recorded. The reaper's two waits are an age floor and then a --- quarantine, and the second one only exists if it was durably written first; without this rule one sweep could --- both propose and execute a collection. +-- * A purge requires a retain_until that was recorded by an EARLIER statement (OLD, not NEW). The reaper's two +-- waits are an age floor and then a quarantine; without this rule one sweep could both propose and execute a +-- collection in a single statement, and the second wait would never have existed at all. -- * A purge is final. Clearing purged_at would claim bytes are back that no one restored. -- +-- The arm matches on the SHAPE of the statement rather than on "one of the two columns changed value", because a +-- sweep that looked at a stream and kept it must be able to record that it looked — advancing the revision and the +-- modification time while both columns keep the values they had. Without that write the claim query would hand the +-- same unreclaimable rows back on every tick, for ever. +-- -- The INSERT arm is tightened for the same reason it rejects a pre-set manifest receipt: a new stream that arrives -- already carrying a retention verdict is not a new stream. -- @@ -76,19 +83,16 @@ BEGIN END IF; -- The SIXTH admissible update shape, and the reason this function is redefined: a RETENTION STATEMENT. It is - -- matched first and returns on its own, because the arms below are written for statements a capturing worker - -- makes and every one of them refuses a terminal row. Both columns are named explicitly, so a statement that - -- touches neither never reaches this arm at all. - IF NEW.retain_until IS DISTINCT FROM OLD.retain_until OR NEW.purged_at IS DISTINCT FROM OLD.purged_at THEN - IF OLD.state = 'Open' THEN - RAISE EXCEPTION 'agent_run_log_stream retention statement rejected on a live stream (id=%); its bytes belong to an open capture session.', OLD.id; - END IF; - IF OLD.purged_at IS NOT NULL THEN - RAISE EXCEPTION 'agent_run_log_stream purge is final; a purged stream admits no further retention statement (id=%).', OLD.id; - END IF; - IF NEW.purged_at IS NOT NULL AND NEW.retain_until IS NULL THEN - RAISE EXCEPTION 'agent_run_log_stream cannot be purged without the retain_until it was quarantined under (id=%).', OLD.id; - END IF; + -- matched first and answers EVERY update to a terminal row, because the arms below are written for statements a + -- capturing worker makes and each one refuses a terminal row outright. The refusal they used to reach is + -- reproduced here word for word, so a caller trying to revive or rewrite a settled stream still reads exactly the + -- message it always did. + -- + -- The shape, not a value change, is what admits it: a retention statement may touch retain_until and purged_at, + -- must advance the revision and the modification time, and may touch NOTHING else. Matching on the shape rather + -- than on "one of the two columns differs" is deliberate — a sweep that looked at a stream and KEPT it has to be + -- able to record that it looked, and that statement changes neither column's value. + IF OLD.state <> 'Open' THEN IF NEW.id IS DISTINCT FROM OLD.id OR NEW.team_id IS DISTINCT FROM OLD.team_id OR NEW.agent_run_id IS DISTINCT FROM OLD.agent_run_id OR NEW.state IS DISTINCT FROM OLD.state OR NEW.stream_kind IS DISTINCT FROM OLD.stream_kind OR NEW.content_type IS DISTINCT FROM OLD.content_type @@ -110,7 +114,15 @@ BEGIN OR NEW.manifest_digest IS DISTINCT FROM OLD.manifest_digest OR NEW.remote_stall_since IS DISTINCT FROM OLD.remote_stall_since OR NEW.remote_stall_code IS DISTINCT FROM OLD.remote_stall_code THEN - RAISE EXCEPTION 'agent_run_log_stream retention statement cannot rewrite anything but its own retention columns (id=%).', OLD.id; + RAISE EXCEPTION 'agent_run_log_stream terminal state is immutable (id=%, state=%).', OLD.id, OLD.state; + END IF; + IF OLD.purged_at IS NOT NULL THEN + RAISE EXCEPTION 'agent_run_log_stream purge is final; a purged stream admits no further retention statement (id=%).', OLD.id; + END IF; + -- OLD, not NEW: the quarantine must have been recorded by an EARLIER statement, or one sweep could both + -- propose a collection and carry it out, and the second of the two waits would never have existed. + IF NEW.purged_at IS NOT NULL AND OLD.retain_until IS NULL THEN + RAISE EXCEPTION 'agent_run_log_stream cannot be purged without the retain_until it was quarantined under (id=%).', OLD.id; END IF; IF NEW.revision <> OLD.revision + 1 OR NEW.last_modified_at < OLD.last_modified_at THEN RAISE EXCEPTION 'agent_run_log_stream revision/time must advance monotonically (id=%, old_revision=%, new_revision=%).', OLD.id, OLD.revision, NEW.revision; @@ -118,6 +130,10 @@ BEGIN RETURN NEW; END IF; + IF NEW.retain_until IS DISTINCT FROM OLD.retain_until OR NEW.purged_at IS DISTINCT FROM OLD.purged_at THEN + RAISE EXCEPTION 'agent_run_log_stream retention statement rejected on a live stream (id=%); its bytes belong to an open capture session.', OLD.id; + END IF; + IF NEW.manifest_digest IS DISTINCT FROM OLD.manifest_digest AND NOT (OLD.state = 'Open' AND NEW.state = 'Completed' AND NEW.schema_version = 3) THEN RAISE EXCEPTION 'A log manifest receipt may only be set by v3 completion.'; @@ -145,9 +161,6 @@ BEGIN OR NEW.created_at IS DISTINCT FROM OLD.created_at THEN RAISE EXCEPTION 'agent_run_log_stream stable identity is immutable (id=%).', OLD.id; END IF; - IF OLD.state <> 'Open' THEN - RAISE EXCEPTION 'agent_run_log_stream terminal state is immutable (id=%, state=%).', OLD.id, OLD.state; - END IF; IF NEW.revision <> OLD.revision + 1 OR NEW.last_modified_at < OLD.last_modified_at THEN RAISE EXCEPTION 'agent_run_log_stream revision/time must advance monotonically (id=%, old_revision=%, new_revision=%).', OLD.id, OLD.revision, NEW.revision; END IF; diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/AgentRunLogService.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/AgentRunLogService.cs index 8762128a9..4ade8621d 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/AgentRunLogService.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/AgentRunLogService.cs @@ -315,6 +315,10 @@ public async Task ReadRangeAsync(AgentRunLogRangeRequest var requestedEndUnbounded = request.OffsetBytes + request.Length; var snapshot = await ReadSnapshotAsync(request.TeamId, request.StreamId, cancellationToken, request.OffsetBytes, requestedEndUnbounded).ConfigureAwait(false); if (snapshot == null) return RejectRange(AgentRunLogProblemCode.Missing); + // Before the segment walk, and before any range arithmetic: a purged stream's locations are gone, so every + // path below it would end at ArtifactMissing — the code that means bytes vanished when they should not have. + // The reader has to be told which of the two happened, so the tombstone answers first and in its own words. + if (snapshot.Metadata.PurgedAt != null) return RejectRange(AgentRunLogProblemCode.Purged, metadata: snapshot.Metadata); if (request.OffsetBytes > snapshot.Metadata.TotalBytes) return RejectRange(AgentRunLogProblemCode.InvalidRequest, metadata: snapshot.Metadata); var requestedEnd = Math.Min(snapshot.Metadata.TotalBytes, checked(request.OffsetBytes + request.Length)); @@ -623,7 +627,14 @@ private static MemoryStream ReadOnlyStream(ReadOnlyMemory bytes) => private static AgentRunLogOpenResult.Opened Opened(AgentRunLogStream value, bool alreadyOpen, bool reclaimed) => new(Project(value), alreadyOpen, reclaimed) { CaptureSourceBaseOffsetBytes = value.CaptureSourceBaseOffsetBytes, CaptureFinalizedAt = value.CaptureFinalizedAt }; private static AgentRunLogCaptureHead CaptureHead(AgentRunLogStream value) => new(Project(value), value.WorkerFenceEpoch!.Value, value.CaptureSessionId!.Value, value.CaptureSourceBaseOffsetBytes, value.CaptureFinalizedAt); private static AgentRunLogSegmentReceipt Receipt(AgentRunLogSegment value) => new(value.Id, value.SegmentOrdinal, value.StartOffsetBytes, value.LengthBytes, value.SourceStartOffsetBytes, value.SourceLengthBytes, value.ArtifactObjectId); - private static AgentRunLogMetadata Project(AgentRunLogStream value) => new(value.Id, value.AgentRunId, value.StreamKind, value.ContentType, value.ContentEncoding, value.CaptureSource, value.Retention, value.State, value.Revision, value.SegmentCount, value.TotalBytes, value.SourceOffsetBytes, value.ContentDigest == null ? null : Convert.ToHexStringLower(value.ContentDigest), value.CreatedAt, value.LastModifiedAt, value.CompletedAt, value.ErrorCode) { Integrity = ProjectIntegrity(value.SchemaVersion, value.ManifestDigest, value.SegmentCount, value.TotalBytes, value.CompletedAt) }; + private static AgentRunLogMetadata Project(AgentRunLogStream value) => new(value.Id, value.AgentRunId, value.StreamKind, value.ContentType, value.ContentEncoding, value.CaptureSource, value.Retention, value.State, value.Revision, value.SegmentCount, value.TotalBytes, value.SourceOffsetBytes, value.ContentDigest == null ? null : Convert.ToHexStringLower(value.ContentDigest), value.CreatedAt, value.LastModifiedAt, value.CompletedAt, value.ErrorCode) + { + // A purged stream keeps its manifest receipt, and projecting it as an integrity proof would state that these + // bytes were verified — about bytes that are gone. The receipt stays in the row as the record of what WAS + // verified; the projection withholds it, and PurgedAt is what the reader gets instead. + Integrity = value.PurgedAt == null ? ProjectIntegrity(value.SchemaVersion, value.ManifestDigest, value.SegmentCount, value.TotalBytes, value.CompletedAt) : null, + PurgedAt = value.PurgedAt, + }; private static bool SameIdentity(AgentRunLogStream stream, AgentRunLogOpenRequest request) => stream.ContentType == request.ContentType && stream.ContentEncoding == request.ContentEncoding && stream.CaptureSource == request.CaptureSource && stream.Retention == request.Retention && stream.ExpiresAt == request.ExpiresAt; private static bool Valid(AgentRunLogOpenRequest value, DateTimeOffset now) => value.TeamId != Guid.Empty && value.AgentRunId != Guid.Empty && value.WorkerFenceEpoch > 0 && value.CaptureSessionId != Guid.Empty && KeyPattern().IsMatch(value.StreamKind ?? "") && KeyPattern().IsMatch(value.CaptureSource ?? "") && value.ContentType is { Length: <= 255 } && ContentTypePattern().IsMatch(value.ContentType) && (value.ContentEncoding == null || EncodingPattern().IsMatch(value.ContentEncoding)) && Enum.IsDefined(value.Retention) && (value.ExpiresAt == null || value.ExpiresAt > now) && (value.Retention != ArtifactRetention.Ephemeral || value.ExpiresAt != null) && (value.Retention != ArtifactRetention.Permanent || value.ExpiresAt == null); private static bool Valid(AgentRunLogAppendRequest value) => value.TeamId != Guid.Empty && value.AgentRunId != Guid.Empty && value.StreamId != Guid.Empty && value.WorkerFenceEpoch > 0 && value.CaptureSessionId != Guid.Empty && value.ExpectedSegmentOrdinal > 0 && value.ExpectedOffsetBytes >= 0 && value.ExpectedSourceOffsetBytes >= 0 && value.SourceLengthBytes > 0 && value.StorageProfileId != Guid.Empty && value.StorageProfileRevision > 0 && value.ActorId != Guid.Empty && value.Bytes.Length is > 0 and <= MaximumAppendBytes && ValidTimeout(value.OperationTimeout); diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/IAgentRunLogService.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/IAgentRunLogService.cs index 84191d702..431f27486 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/IAgentRunLogService.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/IAgentRunLogService.cs @@ -142,6 +142,14 @@ public sealed record AgentRunLogMetadata( string? ErrorCode) { public AgentRunLogIntegrity? Integrity { get; init; } + + /// + /// When the retention plane reclaimed this stream's bytes, or null while they are still at their destination. It + /// rides on the metadata rather than only on the read result because a caller that LISTS streams never attempts a + /// read: without it, GET /logs would report a reclaimed archive as a completed capture with its byte count + /// intact, which is the precise lie the tombstone exists to prevent. + /// + public DateTimeOffset? PurgedAt { get; init; } } public sealed record AgentRunLogSegmentReceipt(Guid SegmentId, long SegmentOrdinal, long StartOffsetBytes, long LengthBytes, long SourceStartOffsetBytes, long SourceLengthBytes, Guid ArtifactObjectId); @@ -246,4 +254,11 @@ public enum AgentRunLogProblemCode StorageActivationFailed, ProviderTimeout, Unsupported, + + /// + /// The stream settled and its bytes were later reclaimed by the retention plane. Deliberately NOT + /// : that code says an object which should be there is not, and folding a policy + /// reclamation into it would make a healthy deployment indistinguishable from one losing data. + /// + Purged, } diff --git a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs index 64d333886..c3bc4a6ce 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs @@ -58,7 +58,7 @@ public async Task SealAsync(PairedQualificationSealR ExpectedObservationCount = ExpectedCount(protocol, request.Manifest), ObservationCount = observations.Count, QualifiedForCapabilityClaim = outcome.QualifiedForCapabilityClaim, OutcomeJson = outcomeJson, }); - await PinCitedRecordsAsync(protocol.ObservationGroupId, observations, cancellationToken).ConfigureAwait(false); + await PinCitedRecordsAsync(protocol, observations, cancellationToken).ConfigureAwait(false); await _db.SaveChangesAsync(cancellationToken).ConfigureAwait(false); return sealedOutcome with { ResultDigest = resultDigest }; } @@ -72,15 +72,20 @@ public async Task SealAsync(PairedQualificationSealR /// its log streams, its cleanup receipts and the offloaded payloads of its events. Migration 0235 backfills the /// same four closures for results sealed before this existed. /// - private async Task PinCitedRecordsAsync(Guid resultId, IReadOnlyList observations, CancellationToken cancellationToken) + private async Task PinCitedRecordsAsync(PairedQualificationProtocol protocol, IReadOnlyList observations, CancellationToken cancellationToken) { + var resultId = protocol.ObservationGroupId; var runIds = observations.Where(row => row.AgentRunId.HasValue).Select(row => row.AgentRunId!.Value).Distinct().ToList(); if (runIds.Count == 0) return; + // Team-scoped wherever the row carries a team: the run ids come from observations this protocol's own + // validation already bound to its team, and a lookup that ignored the tenant would pin another team's rows + // against this result — which would make THEIR records unreclaimable on the strength of a seal they never saw. + // agent_run_event carries no team column of its own (its parent run does), so it is scoped by run alone. var now = DateTimeOffset.UtcNow; - var streamIds = await _db.AgentRunLogStream.AsNoTracking().Where(stream => runIds.Contains(stream.AgentRunId)).Select(stream => stream.Id).ToListAsync(cancellationToken).ConfigureAwait(false); - var receiptIds = await _db.AgentRunCleanupReceipt.AsNoTracking().Where(receipt => runIds.Contains(receipt.AgentRunId)).Select(receipt => receipt.Id).ToListAsync(cancellationToken).ConfigureAwait(false); + var streamIds = await _db.AgentRunLogStream.AsNoTracking().Where(stream => stream.TeamId == protocol.TeamId && runIds.Contains(stream.AgentRunId)).Select(stream => stream.Id).ToListAsync(cancellationToken).ConfigureAwait(false); + var receiptIds = await _db.AgentRunCleanupReceipt.AsNoTracking().Where(receipt => receipt.TeamId == protocol.TeamId && runIds.Contains(receipt.AgentRunId)).Select(receipt => receipt.Id).ToListAsync(cancellationToken).ConfigureAwait(false); var artifactIds = await _db.AgentRunEvent.AsNoTracking().Where(row => runIds.Contains(row.AgentRunId) && row.DataArtifactId != null).Select(row => row.DataArtifactId!.Value).Distinct().ToListAsync(cancellationToken).ConfigureAwait(false); Pin(DurablePinKind.AgentRun, runIds); diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs index e4bfe98da..f115884f0 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs @@ -16,26 +16,58 @@ namespace CodeSpace.Core.Services.Workflows.Retention.Cursors; /// The bytes go, the head row stays. A purge removes the segment objects' bytes and then stamps /// purged_at on the stream head; the head, its segments and its verification manifest survive. That is /// deliberate: a reader that finds no stream at all cannot tell "reclaimed by policy" from "capture lost it", and the -/// entire point of a retention plane is that reclamation is legible. The Room reads the tombstone and says purged. +/// entire point of a retention plane is that reclamation is legible. The read path answers a purged stream with its +/// own typed code and the Room says purged. /// /// Order, and what a crash leaves. The bytes go first and the tombstone commits last. A crash in between -/// leaves bytes gone and no tombstone — the next sweep re-asks, finds the locations already Purged, and finishes the +/// leaves bytes gone and no tombstone — the next sweep re-asks, finds every location already Purged, and finishes the /// stamp; the drain is idempotent by construction. The opposite order would leave bytes that no surviving row /// remembers how to reach, which is the one outcome nothing can repair. /// -/// What is never collected. A stream a sealed qualification result pins. A stream whose object another -/// stream's segment — or a workflow_artifact row — also names, because the CAS is content-addressed and those -/// are ONE physical object. And anything at all when the question could not be asked: every failure answers -/// , never "unreferenced". +/// What is never collected. Anything still names. And anything at all when +/// the question could not be asked: every failure answers , never +/// "unreferenced". +/// +/// What this cursor cannot reclaim, and says so. A REPLICATED object — one whose bytes sit at more than +/// one destination — cannot be purged at all: IArtifactCasPurgeCoordinator refuses it before touching any +/// location, because deleting every replica needs an object-level claim the schema does not yet represent. This +/// cursor names that case from the locations it has already read rather than from the refusal, since the coordinator +/// rejects it with the same code as "no location at all" and a caller that could not tell them apart would retry a +/// call whose answer can never change. Such a stream is kept and logged under its own message; a deployment that +/// replicates its log storage reclaims nothing here until that claim exists. /// public sealed class LogStreamRetentionCursor : IDurableRetentionCursor, IScopedDependency { /// /// Objects purged per stream per sweep. A 4 GiB capture is thousands of segments, and a sweep that tried to drain - /// one in a single pass would hold a worker for minutes on provider I/O. Partial drains cost nothing: the tombstone - /// is only stamped once every object is gone, and the next sweep continues from whatever is left. + /// one in a single pass would hold a worker for minutes on provider I/O. A partial drain deliberately writes + /// NOTHING, so the stream is re-claimed on the very next tick and continues where it stopped — unlike a refusal, + /// which defers it for the class's recheck interval. + /// + internal const int MaxObjectsPerSweep = 64; + + /// + /// Every place a log stream's CAS object can still be named, as (what, why). This list IS the safety of the + /// purge, exactly as ArtifactReferenceOracle.ReferenceSites is for artifacts: a citer missing from it makes + /// the cursor answer "unreferenced" about bytes something still reaches, which is the one failure mode of this + /// class that destroys data. Public and pinned by a test, so a new citer is a deliberate edit. /// - private const int MaxObjectsPerSweep = 64; + public static readonly IReadOnlyList<(string Table, string Column)> CitationSites = + [ + ("paired_qualification_result_pin", "pinned_id"), + ("agent_run_log_segment", "artifact_object_id"), + ("workflow_artifact", "cas_artifact_object_id"), + ]; + + /// + /// The one column that names an artifact object and is deliberately NOT a citation site, recorded here because + /// leaving it out silently is how a list like this rots. artifact_transfer_intent.artifact_object_id is + /// null on every non-terminal row — ck_artifact_transfer_intent_outcome (migration 0127) requires it — so + /// a saga in flight cannot name the object it is moving, and every row that DOES name one is a finished transfer + /// whose record is history. Probing it would answer "still cited" about every log stream in the deployment for + /// ever, because a transfer is exactly what wrote each segment, and it would look like the cursor working. + /// + public const string CommittedTransfersAreNotCiters = "artifact_transfer_intent.artifact_object_id"; private readonly DbContextOptions _dbOptions; private readonly IArtifactCasPurgeCoordinator _purge; @@ -50,18 +82,35 @@ public LogStreamRetentionCursor(DbContextOptions dbOptions, public DurableRecordClass Class => DurableRecordClass.LogStream; - public async Task> ClaimAsync(DateTimeOffset now, DateTimeOffset terminalBefore, int limit, CancellationToken cancellationToken) + /// + /// The per-team head of the eligible queue. DISTINCT ON (team_id) is materialized first so one tenant with + /// a long backlog cannot starve the others, and the same guards are repeated on the outer select because quals + /// inside a materialized CTE are not re-evaluated against a row another worker settled in between. + /// + public async Task> ClaimAsync(DurableRetentionSweepWindow window, int limit, CancellationToken cancellationToken) { + ArgumentNullException.ThrowIfNull(window); await using var db = CreateDb(); - return await db.AgentRunLogStream.AsNoTracking() - .Where(stream => stream.State != AgentRunLogStreamState.Open && stream.PurgedAt == null - && stream.CompletedAt != null && stream.CompletedAt <= terminalBefore - && (stream.RetainUntil == null || stream.RetainUntil <= now)) - .OrderBy(stream => stream.CompletedAt).ThenBy(stream => stream.Id) - .Take(limit) - .Select(stream => new DurableRetentionCandidate(stream.Id, stream.TeamId, stream.Revision, stream.CompletedAt!.Value, stream.RetainUntil)) - .ToListAsync(cancellationToken).ConfigureAwait(false); + var rows = await db.AgentRunLogStream.FromSqlInterpolated($$""" + WITH fair AS MATERIALIZED ( + SELECT DISTINCT ON (stream.team_id) stream.id, stream.completed_at + FROM agent_run_log_stream stream + WHERE stream.state <> 'Open' AND stream.purged_at IS NULL + AND stream.completed_at IS NOT NULL AND stream.completed_at <= {{window.TerminalBefore}} + AND stream.last_modified_at <= {{window.RecheckBefore}} + ORDER BY stream.team_id, stream.completed_at, stream.id + ) + SELECT stream.*, stream.xmin FROM agent_run_log_stream stream + JOIN fair ON fair.id = stream.id + WHERE stream.state <> 'Open' AND stream.purged_at IS NULL + AND stream.completed_at IS NOT NULL AND stream.completed_at <= {{window.TerminalBefore}} + AND stream.last_modified_at <= {{window.RecheckBefore}} + ORDER BY stream.completed_at, stream.id + LIMIT {{limit}} + """).AsNoTracking().ToListAsync(cancellationToken).ConfigureAwait(false); + + return rows.Select(stream => new DurableRetentionCandidate(stream.Id, stream.TeamId, stream.Revision, stream.CompletedAt!.Value, stream.RetainUntil)).ToArray(); } /// Fail-closed: any failure to probe a citation site answers indeterminate, which every consumer reads as keep. @@ -73,7 +122,7 @@ public async Task ClassifyAsync(DurableRetentionCandida { await using var db = CreateDb(); - if (await IsPinnedAsync(db, candidate.Id, cancellationToken).ConfigureAwait(false)) return DurableReferenceVerdict.Referenced; + if (await IsPinnedAsync(db, candidate, cancellationToken).ConfigureAwait(false)) return DurableReferenceVerdict.Referenced; return await SharesBytesAsync(db, candidate, cancellationToken).ConfigureAwait(false) ? DurableReferenceVerdict.Referenced @@ -93,25 +142,34 @@ public async Task SettleAsync(DurableRetentionCandidate candidate, Durable ArgumentNullException.ThrowIfNull(candidate); ArgumentNullException.ThrowIfNull(decision); - if (decision.Action == DurableRetentionAction.Quarantine) return await StampAsync(candidate, decision.RetainUntil, null, cancellationToken).ConfigureAwait(false); - if (decision.Action != DurableRetentionAction.Collect) return false; - - return await CollectAsync(candidate, decision, cancellationToken).ConfigureAwait(false); + return decision.Action == DurableRetentionAction.Collect + ? await CollectAsync(candidate, decision, cancellationToken).ConfigureAwait(false) + : await StampAsync(candidate, decision.RetainUntil, null, cancellationToken).ConfigureAwait(false); } - /// Bytes first, then the tombstone. A drain that did not finish defers the stream instead of stamping it, so the claim query stops returning it every hour. + /// + /// Bytes first, then the tombstone, and only for a stream this plane can account for. + /// + /// The three non-collecting exits are deliberately different. A drain still making progress writes NOTHING, + /// so the next tick continues it. A refusal defers, so one unreachable destination cannot own a batch slot. And a + /// stream whose bytes are already gone by some OTHER path is never tombstoned: stamping it would make the Room + /// say "purged; retention window elapsed" about a loss this plane did not cause, inverting the very distinction + /// the tombstone exists to preserve. + /// private async Task CollectAsync(DurableRetentionCandidate candidate, DurableRetentionDecision decision, CancellationToken cancellationToken) { - var drained = await PurgeBytesAsync(candidate, cancellationToken).ConfigureAwait(false); - var now = await DatabaseClockAsync(cancellationToken).ConfigureAwait(false); + var drain = await DrainAsync(candidate, cancellationToken).ConfigureAwait(false); - if (!drained) + if (drain == DrainOutcome.Progressing) return false; + + if (drain != DrainOutcome.Drained) { - await StampAsync(candidate, now.Add(DurableRetentionPolicy.LogStream.QuarantineWindow), null, cancellationToken).ConfigureAwait(false); + await DeferAsync(candidate, cancellationToken).ConfigureAwait(false); return false; } + var now = await DatabaseClockAsync(cancellationToken).ConfigureAwait(false); var stamped = await StampAsync(candidate, decision.RetainUntil, now, cancellationToken).ConfigureAwait(false); if (stamped) @@ -121,64 +179,129 @@ private async Task CollectAsync(DurableRetentionCandidate candidate, Durab } /// - /// Removes every object this stream's segments still name, bounded per sweep. True only when NOTHING is left to - /// remove — the tombstone must not claim more than the destination actually gave up. + /// Removes every object this stream's segments still name, bounded per sweep. Only + /// permits a tombstone, and it requires POSITIVE evidence that this plane's own lifecycle emptied the stream: + /// segments were captured, and every object behind them now rests at Purged. /// - private async Task PurgeBytesAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken) + private async Task DrainAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken) { - var objectIds = await ReclaimableObjectsAsync(candidate, MaxObjectsPerSweep + 1, cancellationToken).ConfigureAwait(false); + await using var db = CreateDb(); + var objects = Objects(db, candidate); + + if (!await objects.AnyAsync(cancellationToken).ConfigureAwait(false)) return NothingCaptured(candidate); - if (objectIds.Count == 0) return true; + var reclaimable = await ReclaimableAsync(db, candidate, objects, cancellationToken).ConfigureAwait(false); - foreach (var objectId in objectIds.Take(MaxObjectsPerSweep)) + if (reclaimable.Count == 0) return await DrainedOrLostAsync(db, candidate, objects, cancellationToken).ConfigureAwait(false); + if (reclaimable.FirstOrDefault(row => row.LiveLocations > 1) is { } replicated) return Replicated(candidate, replicated); + + foreach (var row in reclaimable.Take(MaxObjectsPerSweep)) { - if (!await PurgeObjectAsync(candidate, objectId, cancellationToken).ConfigureAwait(false)) return false; + if (!await PurgeObjectAsync(candidate, row.ObjectId, cancellationToken).ConfigureAwait(false)) return DrainOutcome.Refused; } - return objectIds.Count <= MaxObjectsPerSweep; + return reclaimable.Count > MaxObjectsPerSweep ? DrainOutcome.Progressing : DrainOutcome.Drained; } - private async Task PurgeObjectAsync(DurableRetentionCandidate candidate, Guid objectId, CancellationToken cancellationToken) + /// The objects this stream's segments name — the entry point for every question the drain asks. + private static IQueryable Objects(CodeSpaceDbContext db, DurableRetentionCandidate candidate) => + db.AgentRunLogSegment.AsNoTracking() + .Where(segment => segment.TeamId == candidate.TeamId && segment.StreamId == candidate.Id) + .Select(segment => segment.ArtifactObjectId) + .Distinct(); + + /// + /// The next objects whose bytes are still at a destination, each with how many destinations hold it, and one row + /// over the batch so the caller can see whether another sweep is needed. Read inside the collecting pass rather + /// than carried from the decision, so a resumed drain converges instead of re-asking about objects an earlier + /// pass already took. + /// + private static async Task> ReclaimableAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, IQueryable objects, CancellationToken cancellationToken) { - var request = new ArtifactCasPurgeRequest { TeamId = candidate.TeamId, ArtifactObjectId = objectId, ActorId = SystemUsers.SeederId }; - var outcome = await _purge.PurgeAsync(request, cancellationToken).ConfigureAwait(false); + // Projected through an anonymous type rather than straight into the record: a positional constructor inside a + // GROUP BY projection is not translatable, and the exception it raises is swallowed as "could not be + // evaluated" — which reads exactly like a fail-closed keep and hides a cursor that reclaims nothing. + var rows = await db.ArtifactLocation.AsNoTracking() + .Where(location => location.TeamId == candidate.TeamId && objects.Contains(location.ArtifactObjectId) + && location.State != ArtifactLocationState.Purged && location.State != ArtifactLocationState.Deleted) + .GroupBy(location => location.ArtifactObjectId) + .Select(group => new { ObjectId = group.Key, LiveLocations = group.Count() }) + .OrderBy(row => row.ObjectId) + .Take(MaxObjectsPerSweep + 1) + .ToListAsync(cancellationToken).ConfigureAwait(false); - if (outcome is ArtifactCasPurgeResult.Purged) return true; + return rows.Select(row => new SegmentObject(row.ObjectId, row.LiveLocations)).ToArray(); + } - var rejected = (ArtifactCasPurgeResult.Rejected)outcome; - _logger.LogWarning("Log stream {StreamId}: object {ObjectId} was not reclaimed ('{Problem}'); the stream keeps its bytes and its head row", candidate.Id, objectId, rejected.Problem.Code); + /// + /// Nothing is left to remove, so either this plane's own lifecycle emptied the stream or something else did. Only + /// the first may be tombstoned: an object whose locations never reached Purged is a loss this cursor did + /// not cause and must not claim. Counting the unaccounted objects over ALL of them, not the batch, is what keeps + /// a stream with more objects than one sweep can hold from being declared drained on a partial view. + /// + private async Task DrainedOrLostAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, IQueryable objects, CancellationToken cancellationToken) + { + var unaccounted = await objects.CountAsync(objectId => !db.ArtifactLocation + .Any(location => location.TeamId == candidate.TeamId && location.ArtifactObjectId == objectId && location.State == ArtifactLocationState.Purged), cancellationToken).ConfigureAwait(false); - return false; + return unaccounted == 0 ? DrainOutcome.Drained : BytesAlreadyGone(candidate, unaccounted); } /// - /// The objects whose bytes are still somewhere, read inside the collecting pass rather than carried from the - /// decision: a location the previous partial drain already purged is not asked about twice, which is what makes a - /// resumed drain converge. + /// Named here rather than read off a refusal code, because the coordinator cannot tell this refusal from "no + /// location at all" — both reject with ArtifactMissing, and a cursor that guessed between them would keep + /// retrying a call whose answer can never change. The locations this drain already read say it exactly. /// - private async Task> ReclaimableObjectsAsync(DurableRetentionCandidate candidate, int limit, CancellationToken cancellationToken) + private DrainOutcome Replicated(DurableRetentionCandidate candidate, SegmentObject replicated) { - await using var db = CreateDb(); + _logger.LogWarning("Log stream {StreamId}: object {ObjectId} holds bytes at {LocationCount} destinations, which this plane cannot reclaim — deleting every replica needs an object-level purge claim that does not exist yet, so the stream is kept", + candidate.Id, replicated.ObjectId, replicated.LiveLocations); - return await db.AgentRunLogSegment.AsNoTracking() - .Where(segment => segment.TeamId == candidate.TeamId && segment.StreamId == candidate.Id) - .Select(segment => segment.ArtifactObjectId) - .Distinct() - .Where(objectId => db.ArtifactLocation.Any(location => location.TeamId == candidate.TeamId && location.ArtifactObjectId == objectId - && location.State != ArtifactLocationState.Purged && location.State != ArtifactLocationState.Deleted)) - .OrderBy(objectId => objectId) - .Take(limit) - .ToListAsync(cancellationToken).ConfigureAwait(false); + return DrainOutcome.Replicated; + } + + private DrainOutcome NothingCaptured(DurableRetentionCandidate candidate) + { + _logger.LogInformation("Log stream {StreamId}: no segment ever landed, so there are no bytes to reclaim and no purge to record", candidate.Id); + + return DrainOutcome.BytesAlreadyGone; + } + + private DrainOutcome BytesAlreadyGone(DurableRetentionCandidate candidate, int unaccounted) + { + _logger.LogWarning("Log stream {StreamId}: {UnaccountedObjects} of its segment objects hold no reclaimable location and were never purged by this plane; the head row is NOT tombstoned, so the loss is not reported as a retention purge", + candidate.Id, unaccounted); + + return DrainOutcome.BytesAlreadyGone; } - private static async Task IsPinnedAsync(CodeSpaceDbContext db, Guid streamId, CancellationToken cancellationToken) => + private async Task PurgeObjectAsync(DurableRetentionCandidate candidate, Guid objectId, CancellationToken cancellationToken) + { + var request = new ArtifactCasPurgeRequest { TeamId = candidate.TeamId, ArtifactObjectId = objectId, ActorId = SystemUsers.SeederId }; + var outcome = await _purge.PurgeAsync(request, cancellationToken).ConfigureAwait(false); + + switch (outcome) + { + case ArtifactCasPurgeResult.Purged: + return true; + case ArtifactCasPurgeResult.Rejected rejected: + _logger.LogWarning("Log stream {StreamId}: object {ObjectId} was not reclaimed ('{Problem}'); the stream keeps its bytes and its head row", candidate.Id, objectId, rejected.Problem.Code); + return false; + default: + _logger.LogWarning("Log stream {StreamId}: object {ObjectId} returned an unrecognised purge outcome; the stream is kept", candidate.Id, objectId); + return false; + } + } + + private static async Task IsPinnedAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, CancellationToken cancellationToken) => await db.PairedQualificationResultPin.AsNoTracking() - .AnyAsync(pin => pin.Kind == DurablePinKind.LogStream && pin.PinnedId == streamId, cancellationToken).ConfigureAwait(false); + .AnyAsync(pin => pin.Kind == DurablePinKind.LogStream && pin.PinnedId == candidate.Id, cancellationToken).ConfigureAwait(false); /// /// Whether anything else holds the same physical bytes. The CAS addresses an object by content, so two streams - /// that captured identical bytes are ONE object with two names, and removing it for one takes the other's content - /// too. Keeping is always safe here; collecting is not. + /// that captured identical bytes — the ordinary case for short or empty output — are ONE object with two names, + /// and removing it for one takes the other's content too. Keeping is always safe here; collecting is not. For why + /// a transfer intent is not asked, see . /// private static async Task SharesBytesAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, CancellationToken cancellationToken) { @@ -193,6 +316,10 @@ private static async Task SharesBytesAsync(CodeSpaceDbContext db, DurableR .AnyAsync(artifact => artifact.CasArtifactObjectId != null && objects.Contains(artifact.CasArtifactObjectId.Value), cancellationToken).ConfigureAwait(false); } + /// A keep that could not be settled any other way still has to advance the row's modification time, or the claim query returns it again on the very next tick. + private async Task DeferAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken) => + await StampAsync(candidate, candidate.RetainUntil, null, cancellationToken).ConfigureAwait(false); + /// /// The one write this cursor makes, and the only shape migration 0236's guard admits: the two retention columns, /// a revision that advances, nothing else. The revision is also the fence — a stream another worker moved since @@ -222,4 +349,26 @@ private async Task DatabaseClockAsync(CancellationToken cancella return await db.Database.SqlQueryRaw("SELECT clock_timestamp() AS \"Value\"").SingleAsync(cancellationToken).ConfigureAwait(false); } + + /// One segment object as one read saw it: how many destinations still hold its bytes, and whether any location rests at Purged. + private sealed record SegmentObject(Guid ObjectId, int LiveLocations); + + /// What one drain attempt established. Only permits a tombstone. + private enum DrainOutcome + { + /// Every object this stream's segments name is gone through this plane's own purge lifecycle. + Drained, + + /// Objects were removed and more remain. Write nothing: the next tick continues from here. + Progressing, + + /// The destination would not give the bytes up. Defer. + Refused, + + /// The bytes sit at more than one destination, which no claim in this schema can take. Defer, and say so. + Replicated, + + /// Nothing was captured, or the bytes left by a path this plane did not drive. Never a purge. + BytesAlreadyGone, + } } diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs index 26b417c3d..8838a02bb 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs @@ -2,7 +2,7 @@ namespace CodeSpace.Core.Services.Workflows.Retention; -/// What a sweep decided to do with one durable record. +/// What a sweep decided to do with one durable record. Every member except is a keep. public enum DurableRetentionAction { /// First observation of "nothing cites this" — record the quarantine deadline, remove nothing. @@ -11,13 +11,13 @@ public enum DurableRetentionAction /// Both waits have elapsed and nothing cites the record. The ONLY action that removes anything. Collect, - /// Something cites the record. Keep. + /// Something cites the record. Keep, and clear any quarantine the citation invalidated. Referenced, /// The status cannot be established. Keep. Indeterminate, - /// A scheduled wait (an age floor or a quarantine window). Keep, and re-ask when it elapses. + /// A scheduled wait (an age floor or a quarantine window) that has not elapsed. Keep. Wait, } @@ -32,6 +32,11 @@ public sealed record DurableRetentionObservation(DateTimeOffset TerminalAt, Date /// its class's age floor keeps, whatever its citation status. Any verdict other than a definite "nothing cites this" /// keeps. And a first "nothing cites this" observation only ever quarantines — collection needs the quarantine window /// to have elapsed on top of the age floor, which is a second, independent wait. +/// +/// is not advice: it is the value the record's quarantine marker must hold after the +/// settlement. A citation CLEARS it, which is the property that keeps the two waits independent — when a cited record +/// later stops being cited, its quarantine starts again from that observation rather than inheriting a deadline set +/// while something still pointed at it. /// public sealed record DurableRetentionDecision(DurableRetentionAction Action, string? Code, DateTimeOffset? RetainUntil) { @@ -44,15 +49,15 @@ public static DurableRetentionDecision Decide(DurableRetentionRule? rule, Durabl { ArgumentNullException.ThrowIfNull(observation); - if (rule is null) return Indeterminate("retention-class-unregistered"); + if (rule is null) return Indeterminate("retention-class-unregistered", observation.RetainUntil); var eligibleAt = observation.TerminalAt.Add(rule.MinimumAge); - if (observation.Now < eligibleAt) return Wait("age-floor-open", eligibleAt); + if (observation.Now < eligibleAt) return Wait("age-floor-open", observation.RetainUntil); if (observation.Verdict == DurableReferenceVerdict.Referenced) return Referenced(); - if (observation.Verdict != DurableReferenceVerdict.Unreferenced) return Indeterminate("reference-status-indeterminate"); + if (observation.Verdict != DurableReferenceVerdict.Unreferenced) return Indeterminate("reference-status-indeterminate", observation.RetainUntil); if (observation.RetainUntil is not { } retainUntil) return Quarantine(observation.Now.Add(rule.QuarantineWindow)); @@ -61,7 +66,10 @@ public static DurableRetentionDecision Decide(DurableRetentionRule? rule, Durabl public static DurableRetentionDecision Quarantine(DateTimeOffset retainUntil) => new(DurableRetentionAction.Quarantine, null, retainUntil); public static DurableRetentionDecision Collect(DateTimeOffset retainUntil) => new(DurableRetentionAction.Collect, null, retainUntil); + + /// A citation. The quarantine marker is cleared, because the observation it recorded has been contradicted. public static DurableRetentionDecision Referenced() => new(DurableRetentionAction.Referenced, null, null); - public static DurableRetentionDecision Indeterminate(string code) => new(DurableRetentionAction.Indeterminate, code, null); - public static DurableRetentionDecision Wait(string code, DateTimeOffset until) => new(DurableRetentionAction.Wait, code, until); + + public static DurableRetentionDecision Indeterminate(string code, DateTimeOffset? retainUntil) => new(DurableRetentionAction.Indeterminate, code, retainUntil); + public static DurableRetentionDecision Wait(string code, DateTimeOffset? retainUntil) => new(DurableRetentionAction.Wait, code, retainUntil); } diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs index c7999af80..146c80fd4 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs @@ -7,46 +7,26 @@ namespace CodeSpace.Core.Services.Workflows.Retention; /// are kept. Values are committed here and changed by a pull request — there is no environment override, because a /// mistyped retention window is unrecoverable data loss and a code review is the control that belongs in front of it. /// +/// One class, because one cursor exists. Every other plane a reaper will eventually reach — cleanup receipts, +/// capture gaps, qualification evidence, budget reservations, terminal transfer intents — gets its rule in the same +/// commit as the cursor that acts on it, so this table can always be read as "what is actually reclaimed". +/// /// An unregistered class is NOT an error a cursor can shrug off: returns null and the decision /// settles Indeterminate, which keeps the record forever. That is what makes removing a class from this table safe. /// public static class DurableRetentionPolicy { - private static readonly TimeSpan Quarantine = TimeSpan.FromHours(24); - /// - /// Thirty days after the stream reached its terminal capture state. Deliberately far longer than any run and - /// longer than any Room a human is still reading: the bytes this reclaims are an archive nobody opened, and the - /// floor costs storage rather than evidence. + /// Thirty days after the stream reached its terminal capture state, then a day of quarantine, and a day between + /// looks at a stream that was kept. Deliberately far longer than any run and longer than any Room a human is still + /// reading: the bytes this reclaims are an archive nobody opened, and the floor costs storage rather than evidence. /// - public static readonly DurableRetentionRule LogStream = new(DurableRecordClass.LogStream, TimeSpan.FromDays(30), Quarantine); - - /// Thirty days after the run went terminal. An Orphaned receipt is NEVER collected — it names a resource nobody reclaimed, so it is kept until something compensates it. - public static readonly DurableRetentionRule CleanupReceipt = new(DurableRecordClass.CleanupReceipt, TimeSpan.FromDays(30), Quarantine); - - /// The same floor as , because a gap describes a span of exactly one stream: outliving it would leave a hole with nothing to be a hole in, and predeceasing it would make an incomplete capture read complete. - public static readonly DurableRetentionRule CaptureGap = new(DurableRecordClass.CaptureGap, TimeSpan.FromDays(30), Quarantine); - - /// Half a year, and only for evidence no sealed result pins. A qualification claim is re-examined long after it was made, so its evidence outlives every other class here by a wide margin. - public static readonly DurableRetentionRule QualificationEvidence = new(DurableRecordClass.QualificationEvidence, TimeSpan.FromDays(180), Quarantine); - - /// Ninety days after reconciliation. A live reservation has no rule at all — it is not in a terminal state and therefore never a candidate. - public static readonly DurableRetentionRule BudgetReservation = new(DurableRecordClass.BudgetReservation, TimeSpan.FromDays(90), Quarantine); - - /// Seven days after the transfer saga ended. The record itself is small; what this class exists to reclaim is the staging object its terminal row still names. - public static readonly DurableRetentionRule TransferIntent = new(DurableRecordClass.TransferIntent, TimeSpan.FromDays(7), Quarantine); + public static readonly DurableRetentionRule LogStream = + new(DurableRecordClass.LogStream, TimeSpan.FromDays(30), TimeSpan.FromHours(24), TimeSpan.FromHours(24)); /// The committed table, public so a test can pin every literal value in it. public static readonly IReadOnlyDictionary Rules = - new Dictionary - { - [LogStream.Class] = LogStream, - [CleanupReceipt.Class] = CleanupReceipt, - [CaptureGap.Class] = CaptureGap, - [QualificationEvidence.Class] = QualificationEvidence, - [BudgetReservation.Class] = BudgetReservation, - [TransferIntent.Class] = TransferIntent, - }; + new Dictionary { [LogStream.Class] = LogStream }; /// The rule for , or null when this build registers none — which every consumer reads as keep. public static DurableRetentionRule? For(DurableRecordClass value) => Rules.TryGetValue(value, out var rule) ? rule : null; diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs index 62d46271f..17f64fb45 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs @@ -21,6 +21,13 @@ namespace CodeSpace.Core.Services.Workflows.Retention; /// tombstone is a conditional update on the record's revision. Two workers sweeping the same record at the same time /// therefore remove the bytes at most once, and exactly one of them writes the tombstone; the loser settles nothing /// and the next sweep meets a row it never held. +/// +/// Why every decision is settled, including a keep. A record nothing can ever reclaim — one a sealed +/// result pins for good, one whose bytes another record also names — is claimed by the oldest-first query on every +/// tick. If a keep wrote nothing, a few hundred such records deployment-wide would fill the batch for ever and the +/// sweep would report a healthy-looking claim count while collecting nothing, never reaching the records behind +/// them. So a keep is a settlement too: it advances the record's own modification time, and the claim query leaves +/// it alone for the class's recheck interval. /// public sealed class DurableRetentionReaper : IDurableRetentionReaper { @@ -62,7 +69,8 @@ private async Task SweepCursorAsync(IDurableRetentionCursor cursor, DateTimeOffs return; } - var candidates = await cursor.ClaimAsync(now, now.Subtract(rule.MinimumAge), BatchSize, cancellationToken).ConfigureAwait(false); + var window = new DurableRetentionSweepWindow(now, now.Subtract(rule.MinimumAge), now.Subtract(rule.RecheckInterval)); + var candidates = await cursor.ClaimAsync(window, BatchSize, cancellationToken).ConfigureAwait(false); counts.Claimed += candidates.Count; foreach (var candidate in candidates) @@ -92,16 +100,16 @@ private async Task SweepCandidateAsync(IDurableRetention } } - /// A settlement the cursor could not complete — a lost race, a refused removal, a drain that needs another sweep — is reported as the keep it is. + /// + /// Every decision is handed to the cursor, including the keeps: a keep that wrote nothing would be re-claimed on + /// every tick for ever. A settlement the cursor could not complete — a lost race, a refused removal, a drain that + /// needs another sweep — is reported as the keep it is. + /// private static async Task ApplyAsync(IDurableRetentionCursor cursor, DurableRetentionCandidate candidate, - DurableRetentionDecision decision, CancellationToken cancellationToken) - { - if (decision.Action is not (DurableRetentionAction.Quarantine or DurableRetentionAction.Collect)) return decision.Action; - - return await cursor.SettleAsync(candidate, decision, cancellationToken).ConfigureAwait(false) + DurableRetentionDecision decision, CancellationToken cancellationToken) => + await cursor.SettleAsync(candidate, decision, cancellationToken).ConfigureAwait(false) ? decision.Action : DurableRetentionAction.Indeterminate; - } /// The database's clock, not the worker's: every deadline these decisions compare against was written by a database clock too. private async Task DatabaseClockAsync(CancellationToken cancellationToken) @@ -117,22 +125,19 @@ private sealed class SweepCounts private int Quarantined { get; set; } private int Collected { get; set; } private int Referenced { get; set; } - private int Indeterminate { get; set; } - private int Waiting { get; set; } + private int Kept { get; set; } public void Record(DurableRetentionAction action) { if (action == DurableRetentionAction.Quarantine) Quarantined++; else if (action == DurableRetentionAction.Collect) Collected++; else if (action == DurableRetentionAction.Referenced) Referenced++; - else if (action == DurableRetentionAction.Wait) Waiting++; - else Indeterminate++; + else Kept++; } public DurableRetentionSweepSummary Summary() => new() { - Claimed = Claimed, Quarantined = Quarantined, Collected = Collected, - Referenced = Referenced, Indeterminate = Indeterminate, Waiting = Waiting, + Claimed = Claimed, Quarantined = Quarantined, Collected = Collected, Referenced = Referenced, Kept = Kept, }; } } diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs index 7fba6588a..847f002b5 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs @@ -18,36 +18,50 @@ public enum DurableReferenceVerdict /// /// One record a sweep is considering, in the only terms the generic loop needs: who it is, when its retention clock /// started, and what an earlier sweep already decided about it. Plane-neutral on purpose — the cursor keeps whatever -/// else it needs to settle the row. +/// else it needs to settle the row. is the fence: a settlement applies only while the +/// record still holds the revision the claim saw. /// public sealed record DurableRetentionCandidate(Guid Id, Guid TeamId, long Revision, DateTimeOffset TerminalAt, DateTimeOffset? RetainUntil); +/// +/// The three instants one sweep measures everything against, computed once by the loop from the class's own rule so a +/// cursor cannot widen its candidate set past the policy. +/// +/// The database clock at sweep start. Every deadline compared against it was written by a database clock too. +/// The age floor: a record that went terminal after this is not a candidate at all. +/// The deferral: a record last settled after this was already looked at recently and is left alone. +public sealed record DurableRetentionSweepWindow(DateTimeOffset Now, DateTimeOffset TerminalBefore, DateTimeOffset RecheckBefore); + /// /// One plane's answers to the three questions the reaper asks: which records are candidates, does anything still cite /// this one, and apply this decision. Deliberately narrow (Rule 7): no policy, no clock, no batching — those belong to /// the loop, so every plane inherits the same waits rather than re-deriving them. /// /// is fail-closed by contract: any failure to reach a citation site answers -/// , never . -/// returns false when the record moved under the sweep, which costs nothing — the next -/// sweep meets a row this one never held. +/// , never . /// public interface IDurableRetentionCursor { DurableRecordClass Class { get; } /// - /// Records in a terminal state that went terminal at or before (the class's age - /// floor, computed by the loop), are not already collected, and are not deferred past . - /// Oldest first, at most . + /// Records in a terminal state that went terminal at or before , + /// are not already collected, and were not settled since . + /// Oldest first, fairly across tenants, at most . /// - /// The deferral half is what keeps one unreclaimable record from owning a batch slot forever: a sweep that - /// cannot finish pushes the record's own retention deadline forward, and this query then skips it until then. + /// The recheck half is what keeps one unreclaimable record from owning a batch slot for good: every + /// settlement — including a keep — advances the record's own modification time, and this query then leaves it + /// alone until the interval elapses. /// - Task> ClaimAsync(DateTimeOffset now, DateTimeOffset terminalBefore, int limit, CancellationToken cancellationToken); + Task> ClaimAsync(DurableRetentionSweepWindow window, int limit, CancellationToken cancellationToken); Task ClassifyAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken); - /// Applies . Only and write anything; every other action is a keep with nothing to record. + /// + /// Applies under the candidate's revision fence. False means nothing was settled — + /// the record moved under this sweep, or the work it needed could not be finished — and the loop reports it as a + /// keep. A cursor decides for itself whether an unfinished settlement also defers the record; work that is making + /// progress must NOT, or a drain that needs several passes would take one recheck interval per pass. + /// Task SettleAsync(DurableRetentionCandidate candidate, DurableRetentionDecision decision, CancellationToken cancellationToken); } diff --git a/backend/src/CodeSpace.Messages/Dtos/Agents/AgentRunLogDtos.cs b/backend/src/CodeSpace.Messages/Dtos/Agents/AgentRunLogDtos.cs index 356164e35..ed5b1ee56 100644 --- a/backend/src/CodeSpace.Messages/Dtos/Agents/AgentRunLogDtos.cs +++ b/backend/src/CodeSpace.Messages/Dtos/Agents/AgentRunLogDtos.cs @@ -73,6 +73,9 @@ public enum AgentRunLogReadAvailability AccessDenied, ProviderTimeout, Unsupported, + + /// The capture settled and the retention plane later reclaimed its bytes. Gone, and gone on purpose — which is exactly what does not say. + Purged, } public sealed record AgentRunLogReadProblem diff --git a/backend/src/CodeSpace.Messages/Retention/DurableRetention.cs b/backend/src/CodeSpace.Messages/Retention/DurableRetention.cs index 04d5637a4..b05b57fec 100644 --- a/backend/src/CodeSpace.Messages/Retention/DurableRetention.cs +++ b/backend/src/CodeSpace.Messages/Retention/DurableRetention.cs @@ -1,39 +1,30 @@ namespace CodeSpace.Messages.Retention; /// -/// A class of durable record that has a retention rule. Membership is the same test the artifact plane applies to -/// ArtifactRetentionClass: a class exists only when the COMPLETE set of places that can still cite one of its -/// records is enumerable — a column, or a pin written in the citing statement's own transaction. A record whose -/// citations can also reach a JSON payload has no class and is therefore never a reap candidate. +/// A class of durable record that has a retention rule AND a cursor that can act on it. A class is added here together +/// with the cursor that sweeps it, never ahead of one: a rule with no cursor reclaims nothing and reads, to anyone +/// scanning this file, as a promise the system does not keep. +/// +/// Membership is the same test the artifact plane applies to ArtifactRetentionClass: a class exists only +/// when the COMPLETE set of places that can still cite one of its records is enumerable — a column, or a pin written in +/// the citing statement's own transaction. A record whose citations can also reach a JSON payload has no class and is +/// therefore never a reap candidate. /// public enum DurableRecordClass { /// An agent_run_log_stream head and the routed CAS bytes its segments name. LogStream = 1, - - /// An agent_run_cleanup_receipt row (migration 0229). - CleanupReceipt = 2, - - /// A workflow_run_capture_gap row (migration 0231). Kept exactly as long as the stream whose missing span it describes. - CaptureGap = 3, - - /// The paired-qualification evidence tables (migrations 0218–0225) behind a sealed result. - QualificationEvidence = 4, - - /// A budget_reservation row (migration 0104) that has already been reconciled. - BudgetReservation = 5, - - /// An artifact_transfer_intent row (migration 0226) that reached a terminal state still holding a staging key. - TransferIntent = 6, } /// /// One class's committed rule. is the age floor measured from the record's own terminal /// instant: below it the record is not even considered, so a citation that is still in flight cannot be outrun. /// is the second, independent wait measured from the first observation of -/// "nothing cites this" — collection needs BOTH to have elapsed. +/// "nothing cites this" — collection needs BOTH to have elapsed. is neither: it is +/// how long a record that was looked at and KEPT is left alone, so one record nothing can ever reclaim cannot own a +/// batch slot for good. /// -public sealed record DurableRetentionRule(DurableRecordClass Class, TimeSpan MinimumAge, TimeSpan QuarantineWindow); +public sealed record DurableRetentionRule(DurableRecordClass Class, TimeSpan MinimumAge, TimeSpan QuarantineWindow, TimeSpan RecheckInterval); /// What one bounded sweep did. Every claimed record lands in exactly one of these buckets. public sealed record DurableRetentionSweepSummary @@ -46,15 +37,19 @@ public sealed record DurableRetentionSweepSummary /// Both waits elapsed with no citation. The only bucket that removed anything. public required int Collected { get; init; } - /// Something still cites the record. Kept. + /// Something still cites the record. Kept, and not looked at again until the recheck interval elapses. public required int Referenced { get; init; } - /// The question could not be answered, or the removal could not be completed. Kept. - public required int Indeterminate { get; init; } - - /// A scheduled wait — an age floor or a quarantine window that has not elapsed. Kept, and re-asked next sweep. - public required int Waiting { get; init; } + /// + /// A keep this sweep could not turn into a quarantine or a collection: an unanswered citation question, a refused + /// removal, a drain that needs another pass, or a record whose bytes were already gone by some other path. Kept. + /// + /// Distinguishing this from is the point of having it: a sweep that reports Claimed + /// and nothing else is a sweep that is re-asking, and an operator reading these numbers has to be able to see + /// that rather than infer healthy work from a non-zero claim count. + /// + public required int Kept { get; init; } public static DurableRetentionSweepSummary Empty { get; } = - new() { Claimed = 0, Quarantined = 0, Collected = 0, Referenced = 0, Indeterminate = 0, Waiting = 0 }; + new() { Claimed = 0, Quarantined = 0, Collected = 0, Referenced = 0, Kept = 0 }; } diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs index 9f800edae..1849834e1 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs @@ -16,6 +16,7 @@ using CodeSpace.Messages.Artifacts; using CodeSpace.Messages.Dtos.Sessions.Room; using CodeSpace.Messages.Enums; +using CodeSpace.Messages.Retention; using Microsoft.EntityFrameworkCore; using Microsoft.Extensions.Logging.Abstractions; using Npgsql; @@ -29,14 +30,21 @@ namespace CodeSpace.IntegrationTests.Workflows; /// rows a seal writes in its own transaction, and the order in which bytes and their tombstone commit. /// /// The positive control at the top is what proves the counter-examples below it are not passing vacuously: each -/// of those builds a stream that is ONE property away from collectable and asserts its bytes survive. +/// of those builds a stream that is ONE property away from collectable and asserts its bytes survive. Every assertion +/// is about a row this class created — never a sweep tally, which is deployment-wide over a shared database and +/// therefore counts other classes' rows as readily as its own. +/// +/// The class also reclaims what it staged (): a bounded global sweep elsewhere in the +/// suite — the artifact-location verifier — selects the hundred least-recently-verified placements across every team, +/// so a class that leaves live placements behind a destination it then deletes changes what that sweep sees. /// [Collection(PostgresCollection.Name)] [Trait("Category", "Integration")] -public sealed class DurableRetentionReaperFlowTests : IDisposable +public sealed class DurableRetentionReaperFlowTests : IAsyncLifetime { private readonly PostgresFixture _fixture; private readonly List _roots = []; + private readonly List _staged = []; public DurableRetentionReaperFlowTests(PostgresFixture fixture) => _fixture = fixture; @@ -50,29 +58,52 @@ public async Task A_terminal_unpinned_log_stream_past_its_rule_is_purged_blobs_f { var world = await SeedWorldAsync(); var stream = await CaptureAsync(world, "the archived transcript of a run nobody opened again"); - await AgeTerminalAsync(stream, TimeSpan.FromDays(31)); + await AgeAsync(stream, TimeSpan.FromDays(31)); - var first = await SweepAsync(); + await SweepAsync(); - first.Quarantined.ShouldBeGreaterThanOrEqualTo(1, "the first uncited observation only starts the quarantine clock"); - (await StreamAsync(stream)).PurgedAt.ShouldBeNull("the sweep that first noticed the stream must not touch its bytes"); - (await StreamAsync(stream)).RetainUntil.ShouldNotBeNull("the quarantine deadline has to be durable — an in-memory one is no wait at all"); + var quarantined = await StreamAsync(stream); + quarantined.PurgedAt.ShouldBeNull("the sweep that first noticed the stream must not touch its bytes"); + quarantined.RetainUntil.ShouldNotBeNull("the quarantine deadline has to be durable — an in-memory one is no wait at all"); (await BytesReadableAsync(world, stream)).ShouldBeTrue(); await ElapseQuarantineAsync(stream); - var second = await SweepAsync(); + await SweepAsync(); - second.Collected.ShouldBeGreaterThanOrEqualTo(1); var purged = await StreamAsync(stream); purged.PurgedAt.ShouldNotBeNull("the head row is the tombstone: it outlives its bytes so a reader is told they were reclaimed"); purged.State.ShouldBe(AgentRunLogStreamState.Completed, "a purge is not a capture verdict — the state it settled in is untouched"); purged.TotalBytes.ShouldBeGreaterThan(0, "the byte head still records what was captured; only the bytes themselves are gone"); (await UnpurgedLocationsAsync(world, stream)).ShouldBe(0, "every location behind the stream's segments must be drained before the tombstone commits"); - (await BytesReadableAsync(world, stream)).ShouldBeFalse("the destination no longer holds the object"); RoomProjector.SummarizeLogs([Row(purged)]).Status.ShouldBe(RoomAgentLogStatus.Purged, "the Room must say purged rather than claim an integrity proof over bytes that are gone"); } + /// + /// What a reader is told about a stream whose bytes were reclaimed. A 410 alone is the answer a LOST object gets, + /// so the code beside it carries the distinction, and the metadata must stop advertising an integrity proof over + /// bytes that are gone. + /// + [Fact] + public async Task The_read_path_answers_a_purged_stream_as_purged_and_claims_no_integrity() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "bytes a reader will come looking for"); + var before = await MetadataAsync(world, stream); + before.Integrity.ShouldNotBeNull().ManifestDigest.ShouldNotBeNull("the positive control: while the bytes are there the stream does carry its proof"); + before.PurgedAt.ShouldBeNull(); + + await PurgeAsync(world, stream); + + var read = (await ReadAsync(world, stream)).ShouldBeOfType(); + read.Problem.Code.ShouldBe(AgentRunLogProblemCode.Purged, + "ArtifactMissing means an object that should be there is not; a policy reclamation must never be reported in the vocabulary of data loss"); + var after = await MetadataAsync(world, stream); + after.PurgedAt.ShouldNotBeNull("a caller that LISTS streams never attempts a read, so the tombstone has to ride on the metadata too"); + after.Integrity.ShouldBeNull("the manifest receipt is still in the row as the record of what WAS verified; projecting it now would claim these bytes are verifiable"); + after.TotalBytes.ShouldBe(before.TotalBytes, "what was captured is still stated; only the claim that it can be read is withdrawn"); + } + /// /// The pin-table-first property, from the other side: a stream a sealed qualification result cites is never /// collected, however long both windows have been open. Mutation: drop the pin probe in @@ -84,11 +115,11 @@ public async Task A_pinned_log_stream_past_its_rule_survives() var world = await SeedWorldAsync(); var stream = await CaptureAsync(world, "evidence a capability claim was computed from"); await SealResultAsync(world); - await AgeTerminalAsync(stream, TimeSpan.FromDays(400)); + await AgeAsync(stream, TimeSpan.FromDays(400)); await SweepAsync(); - (await StreamAsync(stream)).RetainUntil.ShouldBeNull("a cited stream is never even quarantined — the verdict is keep, not a wait"); + (await StreamAsync(stream)).RetainUntil.ShouldBeNull("a cited stream is never quarantined — the verdict is keep, not a wait"); // The strongest form of the claim: a quarantine deadline that has ALREADY elapsed, so the only thing left // between the reaper and the bytes is the pin. @@ -97,10 +128,37 @@ public async Task A_pinned_log_stream_past_its_rule_survives() var kept = await StreamAsync(stream); kept.PurgedAt.ShouldBeNull("a sealed result still cites this stream; no elapsed window outranks that"); + kept.RetainUntil.ShouldBeNull("and the citation cleared the quarantine it contradicts, so an un-pinned stream starts its wait afresh"); (await BytesReadableAsync(world, stream)).ShouldBeTrue(); (await UnpurgedLocationsAsync(world, stream)).ShouldBeGreaterThan(0); } + /// + /// The starvation property. A record nothing can ever reclaim is claimed by an oldest-first query on every tick, + /// so a keep that wrote nothing would fill the batch with the same rows for ever and the sweep would report a + /// healthy claim count while collecting nothing. Mutation: stop settling the keeps (or drop the recheck predicate + /// from the claim query) and the kept stream comes back in the second claim, ahead of the eligible one. + /// + [Fact] + public async Task A_stream_that_was_kept_is_not_re_claimed_until_its_recheck_interval_elapses() + { + var world = await SeedWorldAsync(); + var pinned = await CaptureAsync(world, "evidence pinned for ever"); + await SealResultAsync(world); + await AgeAsync(pinned, TimeSpan.FromDays(400)); + await SweepAsync(); + + // A second team, because the claim is fair per tenant: two streams of ONE team can never be in the same + // batch, so a single-team version of this test could not tell starvation from that fairness. + var neighbour = await SeedWorldAsync(); + var collectable = await CaptureAsync(neighbour, "a younger archive behind it"); + await AgeAsync(collectable, TimeSpan.FromDays(31)); + var claimed = await ClaimAsync(limit: 200); + + claimed.ShouldNotContain(pinned, "a stream this sweep already looked at and kept must not be handed back on the next tick — at scale it would own the batch for ever"); + claimed.ShouldContain(collectable, "and the stream behind it, which no sweep has looked at, must be reachable"); + } + /// /// The atomicity claim, measured rather than asserted: xmin is the id of the transaction that inserted a /// row, so a result and its pins sharing one xmin IS "written in the same transaction". Mutation: move the @@ -132,40 +190,245 @@ public async Task A_sealed_result_pins_every_stream_and_artifact_it_cites_in_one } /// - /// Bytes before rows, proven by the failure: with the destination gone the purge is refused, and the sweep must - /// leave the head row exactly as it found it. Mutation: stamp the tombstone before (or without) the byte removal - /// and this reds — the Room would then read "purged" about an archive still sitting at the destination, and no - /// later sweep would ever reclaim it, because a purged row is never claimed again. + /// Migration 0235's backfill and the writer have to agree, because a result sealed before this plane existed would + /// otherwise read as citing nothing and its evidence would become collectable. Mutation: change either closure + /// (drop the cleanup receipts from the backfill, add a kind to the writer) and the two sets differ here. + /// + [Fact] + public async Task The_backfill_writes_exactly_the_pins_the_seal_writes() + { + var world = await SeedWorldAsync(); + await CaptureAsync(world, "a stream a pre-existing result cites"); + await SeedCitedRecordsAsync(world); + var groupId = await SealResultAsync(world); + var written = (await PinsAsync(groupId)).Select(Identity).ToList(); + + written.ShouldNotBeEmpty("the comparison below is vacuous unless the writer actually pinned something"); + await ForgetPinsAsync(groupId); + await RunBackfillAsync(); + + (await PinsAsync(groupId)).Select(Identity).Order(StringComparer.Ordinal) + .ShouldBe(written.Order(StringComparer.Ordinal), + "0235's backfill is the same four closures as PairedQualificationResultStore.PinCitedRecordsAsync; if they drift, results sealed before this migration lose the protection new ones get"); + } + + /// + /// Bytes before rows, proven by the failure: with the destination refusing, the sweep must leave the head row's + /// tombstone unwritten. Mutation: stamp the tombstone before (or without) the byte removal and this reds — the + /// Room would then read "purged" about an archive still sitting at the destination, and no later sweep would ever + /// reclaim it, because a purged row is never claimed again. /// [Fact] - public async Task A_refused_byte_purge_leaves_the_tombstone_unwritten() + public async Task A_refused_byte_purge_leaves_the_tombstone_unwritten_and_defers_the_stream() { var world = await SeedWorldAsync(); var stream = await CaptureAsync(world, "bytes at a destination that stops answering"); - await AgeTerminalAsync(stream, TimeSpan.FromDays(31)); + await AgeAsync(stream, TimeSpan.FromDays(31)); await SweepAsync(); await ElapseQuarantineAsync(stream); + var beforeRefusal = await StreamAsync(stream); - var sweep = await SweepAgainstARefusingDestinationAsync(); + await SweepAgainstARefusingDestinationAsync(); - sweep.Collected.ShouldBe(0, "nothing may be reported collected while the destination refuses to give the bytes up"); var kept = await StreamAsync(stream); kept.PurgedAt.ShouldBeNull("the tombstone must never run ahead of the bytes"); (await UnpurgedLocationsAsync(world, stream)).ShouldBeGreaterThan(0, "a refused delete leaves the location exactly where it was"); - kept.RetainUntil.ShouldNotBeNull().ShouldBeGreaterThan(DateTimeOffset.UtcNow, - "the stream is deferred rather than retried every tick, so one unreachable destination cannot own the batch"); + kept.LastModifiedAt.ShouldBeGreaterThan(beforeRefusal.LastModifiedAt, + "a refusal is deferred rather than retried every tick, so one unreachable destination cannot own a batch slot"); + (await ClaimAsync(limit: 10)).ShouldNotContain(stream); } /// - /// Migration 0236's arm, at the only place that can pin it: the database. Everything the guard refuses here is a - /// statement that would let a retention write masquerade as something else — or let a purge claim a wait it never - /// served. + /// A drain too big for one sweep must NOT be deferred like a refusal: it is making progress, so it writes nothing + /// and the next tick continues it. Mutation: defer a partial drain and a large capture takes one recheck interval + /// per batch of objects — a 4 GiB archive would need months. + /// + [Fact] + public async Task A_drain_too_big_for_one_sweep_writes_nothing_and_the_next_sweep_finishes_it() + { + var world = await SeedWorldAsync(); + var segments = LogStreamRetentionCursor.MaxObjectsPerSweep + 1; + var stream = await CaptureAsync(world, "first", AgentRunLogKinds.StandardOutput, segments); + await AgeAsync(stream, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(stream); + var beforeDrain = await StreamAsync(stream); + + await SweepAsync(); + + var partly = await StreamAsync(stream); + partly.PurgedAt.ShouldBeNull("the tombstone waits until every object is gone"); + partly.LastModifiedAt.ShouldBe(beforeDrain.LastModifiedAt, "a drain that is progressing writes NOTHING, so the very next tick continues it"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(segments - LogStreamRetentionCursor.MaxObjectsPerSweep, "exactly one bounded batch of objects went"); + + await SweepAsync(); + + (await StreamAsync(stream)).PurgedAt.ShouldNotBeNull("the second sweep finishes the drain and stamps the tombstone"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(0); + } + + /// + /// Two streams that captured identical output are ONE content-addressed object under two names — the ordinary case + /// for short or empty logs. Mutation: drop the sibling-segment probe and purging either one silently empties the + /// other, whose head row still says its bytes are there. + /// + [Fact] + public async Task A_stream_whose_bytes_another_stream_also_names_is_never_collected() + { + var world = await SeedWorldAsync(); + var identical = "byte-for-byte the same output"; + var first = await CaptureAsync(world, identical); + var second = await CaptureAsync(world, identical, AgentRunLogKinds.StandardError); + (await ObjectsOfAsync(world, first)).ShouldBe(await ObjectsOfAsync(world, second), "the premise: the CAS stored one object for both captures"); + + await AgeAsync(first, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(first); + await SweepAsync(); + + (await StreamAsync(first)).PurgedAt.ShouldBeNull("another stream still reaches these bytes"); + (await UnpurgedLocationsAsync(world, first)).ShouldBeGreaterThan(0); + (await BytesReadableAsync(world, second)).ShouldBeTrue("and the stream that was never a candidate still reads"); + + // The verdict itself, because the outcome alone does not pin the reason: deduplicated captures also leave one + // object with a placement per write, which the drain refuses separately as replication. Asserting the + // CITATION verdict is what reds when the sibling-segment probe is removed. + var candidate = (await CandidatesAsync(limit: 200)).FirstOrDefault(row => row.Id == first) + ?? new DurableRetentionCandidate(first, world.TeamId, (await StreamAsync(first)).Revision, DateTimeOffset.UtcNow.AddDays(-99), null); + (await Cursor().ClassifyAsync(candidate, CancellationToken.None)).ShouldBe(DurableReferenceVerdict.Referenced, + "the stream is kept because something else NAMES its bytes, not because of how they happen to be placed"); + } + + /// + /// Why a transfer intent is NOT one of the citers the cursor probes, asserted rather than reasoned about: the + /// schema refuses to let a saga in flight name an object at all. Every row that does name one is a finished + /// transfer — and a transfer is what wrote each log segment — so probing the column would answer "still cited" + /// about every log stream in the deployment for ever while looking exactly like a safety property. + /// + [Fact] + public async Task A_transfer_saga_in_flight_cannot_name_the_object_it_is_moving() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "bytes a transfer already moved"); + var objectId = (await ObjectsOfAsync(world, stream)).Single(); + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + db.ArtifactTransferIntent.Add(await InFlightIntentAsync(db, world, objectId)); + + var raised = await Should.ThrowAsync(async () => await db.SaveChangesAsync()); + + PostgresErrorOf(raised).MessageText.ShouldContain("ck_artifact_transfer_intent_outcome", customMessage: + "if a non-terminal intent could ever carry an artifact_object_id, it WOULD be a citer and the cursor would have to probe it"); + (await db.ArtifactTransferIntent.AsNoTracking().CountAsync(intent => intent.ArtifactObjectId == objectId && intent.State != ArtifactTransferState.Committed)) + .ShouldBe(0, "the only rows naming this object are the committed transfer that wrote it"); + } + + /// + /// The age floor, at the CALL SITE rather than in the pure decision: the window the loop computes from the class's + /// rule is what keeps a young stream out of the batch entirely. Mutation: widen the window (or drop the + /// completed_at predicate) and a stream one day short of its floor is claimed. + /// + [Fact] + public async Task A_stream_inside_its_age_floor_is_never_claimed() + { + var world = await SeedWorldAsync(); + var young = await CaptureAsync(world, "a capture from last month, one day short"); + await AgeAsync(young, DurableRetentionPolicy.LogStream.MinimumAge - TimeSpan.FromDays(1)); + + (await ClaimAsync(limit: 50)).ShouldNotContain(young); + + await AgeAsync(young, TimeSpan.FromDays(2)); + + (await ClaimAsync(limit: 50)).ShouldContain(young, "the counter-example is only worth anything if the same stream IS claimed once its floor has passed"); + } + + /// + /// The claim query requires a completion instant because the age floor is measured from one. That predicate can + /// never be the ONLY thing standing between a clockless row and a purge, and this is why: the schema refuses to + /// hold a terminal stream without one at all. Asserting the refusal rather than fabricating the row keeps the + /// guarantee where it actually lives. + /// + [Fact] + public async Task A_terminal_stream_cannot_exist_without_the_instant_its_floor_is_measured_from() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "a settled capture"); + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream DISABLE TRIGGER agent_run_log_stream_enforce_invariants"); + + try + { + var raised = await Should.ThrowAsync(async () => + await db.Database.ExecuteSqlRawAsync("UPDATE agent_run_log_stream SET completed_at = NULL WHERE id = {0}", [stream])); + + PostgresErrorOf(raised).MessageText.ShouldContain("ck_agent_run_log_stream_terminal"); + } + finally + { + await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream ENABLE TRIGGER agent_run_log_stream_enforce_invariants"); + } + } + + /// + /// The fence. A settlement carries the revision its claim saw, and a stream another worker moved in between must + /// take nothing from it. Mutation: drop the revision predicate from the stamp and two workers racing one stream + /// both write, so the loser's stale verdict lands on top of the winner's. + /// + [Fact] + public async Task A_settlement_whose_revision_moved_writes_nothing() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "a stream two workers both claimed"); + await AgeAsync(stream, TimeSpan.FromDays(31)); + var claimed = (await CandidatesAsync(limit: 50)).Single(candidate => candidate.Id == stream); + await ElapseQuarantineAsync(stream); // somebody else moved the row after this claim was taken + var moved = await StreamAsync(stream); + + var settled = await Cursor().SettleAsync(claimed, DurableRetentionDecision.Quarantine(DateTimeOffset.UtcNow.AddDays(1)), CancellationToken.None); + + settled.ShouldBeFalse("the claim is stale, so it settles nothing"); + var after = await StreamAsync(stream); + after.Revision.ShouldBe(moved.Revision, "and it must leave the row exactly where the other worker put it"); + after.RetainUntil.ShouldBe(moved.RetainUntil); + } + + /// + /// The distinction the tombstone exists for. A stream whose bytes were lost by some OTHER path must not be stamped + /// "purged", or the Room would tell the operator a retention window elapsed on an archive nobody reclaimed. + /// Mutation: treat an empty reclaimable set as drained and this reds with a tombstone on a loss this plane did not + /// cause. + /// + [Fact] + public async Task A_stream_whose_bytes_vanished_outside_this_plane_is_never_tombstoned() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "bytes something else lost"); + await AgeAsync(stream, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(stream); + await MarkLocationsDeletedAsync(world, stream); + + await SweepAsync(); + + var kept = await StreamAsync(stream); + kept.PurgedAt.ShouldBeNull("this plane removed nothing, so it may not report a purge"); + RoomProjector.SummarizeLogs([Row(kept)]).Status.ShouldNotBe(RoomAgentLogStatus.Purged, + "\"purged; retention window elapsed\" about a loss nobody chose inverts the one distinction the tombstone preserves"); + } + + /// + /// Migration 0236's arm, at the only place that can pin it: the database. A terminal stream admits a retention + /// statement and NOTHING else — a statement that smuggles anything alongside the two columns reads the same + /// refusal it always did. /// [Theory] + [InlineData("retain_until = now(), state = 'Corrupt'", "terminal state is immutable")] + [InlineData("retain_until = now(), total_bytes = 0", "terminal state is immutable")] + [InlineData("retain_until = now(), completed_at = now()", "terminal state is immutable")] + [InlineData("state = 'Open', completed_at = NULL", "terminal state is immutable")] [InlineData("purged_at = now()", "cannot be purged without the retain_until")] - [InlineData("retain_until = now(), state = 'Corrupt'", "cannot rewrite anything but its own retention columns")] - [InlineData("retain_until = now(), total_bytes = 0", "cannot rewrite anything but its own retention columns")] - [InlineData("retain_until = now(), completed_at = now()", "cannot rewrite anything but its own retention columns")] + [InlineData("purged_at = now(), retain_until = now() - interval '1 day'", "cannot be purged without the retain_until")] public async Task The_guard_refuses_a_retention_statement_that_says_more_than_retention(string smuggled, string refusal) { var world = await SeedWorldAsync(); @@ -205,11 +468,7 @@ public async Task The_guard_refuses_to_un_purge_a_purged_stream() { var world = await SeedWorldAsync(); var stream = await CaptureAsync(world, "a stream that will be reclaimed"); - await AgeTerminalAsync(stream, TimeSpan.FromDays(31)); - await SweepAsync(); - await ElapseQuarantineAsync(stream); - await SweepAsync(); - (await StreamAsync(stream)).PurgedAt.ShouldNotBeNull(); + await PurgeAsync(world, stream); using var scope = _fixture.BeginScope(); var raised = await Should.ThrowAsync(async () => await scope.Resolve().Database.ExecuteSqlRawAsync( @@ -219,96 +478,182 @@ public async Task The_guard_refuses_to_un_purge_a_purged_stream() PostgresErrorOf(raised).MessageText.ShouldContain("purge is final"); } - // ── The world, and the durable stream inside it ──────────────────────────────────────────────────────────────── + /// + /// Why this plane reclaims bytes and not rows, stated as the refusals themselves rather than as prose in a pull + /// request: all three archive tables reject deletion outright, so a tombstone on the surviving head is the only + /// legible way to say the bytes are gone. + /// + [Fact] + public async Task The_log_archive_tables_refuse_every_deletion() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "an archive nothing may delete"); + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + + foreach (var (table, predicate, refusal) in new[] + { + ("agent_run_log_segment", "stream_id", "append-only"), + ("agent_run_log_verification", "stream_id", "durable history"), + ("agent_run_log_stream", "id", "DELETE rejected"), + }) + { + var raised = await Should.ThrowAsync(async () => + await db.Database.ExecuteSqlRawAsync($"DELETE FROM {table} WHERE {predicate} = {{0}}", [stream])); + + PostgresErrorOf(raised).MessageText.ShouldContain(refusal, customMessage: $"{table} is supposed to refuse deletion; if it no longer does, this plane could reclaim rows instead of tombstoning them"); + } + } + + // ── The world, and the durable streams inside it ─────────────────────────────────────────────────────────────── - /// One real capture through the real service: an open, one segment at a real local-rwx destination, a finalized source and a v3 completion. - private async Task CaptureAsync(World world, string text) + /// One real capture through the real service: an open, segments at a real local-rwx destination, a finalized source and a v3 completion. + private async Task CaptureAsync(World world, string text, string kind = AgentRunLogKinds.StandardOutput, int segments = 1) { using var scope = _fixture.BeginScope(); var logs = Logs(scope); var session = Guid.NewGuid(); - var bytes = System.Text.Encoding.UTF8.GetBytes(text); - var opened = (await logs.OpenAsync(Open(world, session), CancellationToken.None)).ShouldBeOfType(); - var appended = (await logs.AppendAsync(new AgentRunLogAppendRequest + var opened = (await logs.OpenAsync(Open(world, session, kind), CancellationToken.None)).ShouldBeOfType(); + var metadata = opened.Metadata; + + for (var ordinal = 1; ordinal <= segments; ordinal++) { - TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = opened.Metadata.StreamId, WorkerFenceEpoch = Fence, - CaptureSessionId = session, ExpectedSegmentOrdinal = 1, ExpectedOffsetBytes = 0, ExpectedSourceOffsetBytes = 0, - SourceLengthBytes = bytes.Length, StorageProfileId = world.StorageProfileId, StorageProfileRevision = 1, - ActorId = world.ActorId, Bytes = bytes, - }, CancellationToken.None)).ShouldBeOfType(); + // Distinct bytes per segment on purpose: the CAS is content-addressed, so repeating a payload would make + // two segments one object and the drain would have fewer objects to take than the test counted on. + var bytes = System.Text.Encoding.UTF8.GetBytes($"{text} #{ordinal:D4} {new string('x', 24)}"); + metadata = (await logs.AppendAsync(new AgentRunLogAppendRequest + { + TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = metadata.StreamId, WorkerFenceEpoch = Fence, + CaptureSessionId = session, ExpectedSegmentOrdinal = ordinal, ExpectedOffsetBytes = metadata.TotalBytes, + ExpectedSourceOffsetBytes = metadata.SourceOffsetBytes, SourceLengthBytes = bytes.Length, + StorageProfileId = world.StorageProfileId, StorageProfileRevision = 1, ActorId = world.ActorId, Bytes = bytes, + }, CancellationToken.None)).ShouldBeOfType().Metadata; + } + var finalized = (await logs.FinalizeSourceAsync(new AgentRunLogFinalizeSourceRequest { - TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = opened.Metadata.StreamId, WorkerFenceEpoch = Fence, - CaptureSessionId = session, ExpectedRevision = appended.Metadata.Revision, ExpectedSourceOffsetBytes = appended.Metadata.SourceOffsetBytes, + TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = metadata.StreamId, WorkerFenceEpoch = Fence, + CaptureSessionId = session, ExpectedRevision = metadata.Revision, ExpectedSourceOffsetBytes = metadata.SourceOffsetBytes, }, CancellationToken.None)).ShouldBeOfType(); - (await logs.CompleteAsync(new AgentRunLogCompleteRequest + // Completion verifies a bounded number of segments per call, so a many-segment capture reports Progress until + // its manifest is whole — exactly what the production bridge does, and what a single call would silently miss. + var revision = finalized.Metadata.Revision; + + for (var pass = 0; pass < segments + 2; pass++) { - TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = opened.Metadata.StreamId, WorkerFenceEpoch = Fence, - CaptureSessionId = session, ExpectedRevision = finalized.Metadata.Revision, - }, CancellationToken.None)).ShouldBeOfType(); + var completion = await logs.CompleteAsync(new AgentRunLogCompleteRequest + { + TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = metadata.StreamId, WorkerFenceEpoch = Fence, + CaptureSessionId = session, ExpectedRevision = revision, + }, CancellationToken.None); - return opened.Metadata.StreamId; + if (completion is AgentRunLogCompleteResult.Completed) break; + + revision = completion.ShouldBeOfType().Metadata.Revision; + } + + (await StreamAsync(metadata.StreamId)).State.ShouldBe(AgentRunLogStreamState.Completed, "the capture this test starts from has to settle, or every assertion after it is about the wrong row"); + _staged.Add(new StagedStream(world.TeamId, metadata.StreamId)); + + return metadata.StreamId; } /// - /// Time travel for the age floor only. The trigger is suspended for this one statement because the guard admits - /// no rewrite of completed_at at all — which is exactly the property the theory above pins — and the floor - /// is measured in days that no test can wait out. + /// Time travel for the two clocks the claim query reads: how long ago the capture settled, and how long ago + /// anything last touched the row. Every timestamp on the row moves by the SAME delta, so their ordering — which + /// ck_agent_run_log_stream_time enforces and which is a real invariant rather than test scaffolding — still + /// holds; the shift is at least the recheck interval so that "age it by nothing" still reopens the deferral gate. + /// + /// The trigger is suspended for this one statement because the guard admits no rewrite of + /// completed_at at all — which is exactly the property the guard theory pins — and both windows are + /// measured in days no test can wait out. /// - private async Task AgeTerminalAsync(Guid streamId, TimeSpan age) + private async Task AgeAsync(Guid streamId, TimeSpan age) { + var shift = age > TimeSpan.FromDays(2) ? age : TimeSpan.FromDays(2); + const string sql = "UPDATE agent_run_log_stream SET created_at = created_at - {0}::interval, capture_finalized_at = capture_finalized_at - {0}::interval, " + + "completed_at = completed_at - {0}::interval, last_modified_at = last_modified_at - {0}::interval WHERE id = {1}"; using var scope = _fixture.BeginScope(); var db = scope.Resolve(); await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream DISABLE TRIGGER agent_run_log_stream_enforce_invariants"); - try - { - await db.Database.ExecuteSqlRawAsync("UPDATE agent_run_log_stream SET completed_at = completed_at - {0}::interval WHERE id = {1}", - [$"{age.TotalSeconds} seconds", streamId]); - } - finally - { - await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream ENABLE TRIGGER agent_run_log_stream_enforce_invariants"); - } + try { await db.Database.ExecuteSqlRawAsync(sql, [$"{shift.TotalSeconds} seconds", streamId]); } + finally { await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream ENABLE TRIGGER agent_run_log_stream_enforce_invariants"); } } - /// Pulls the quarantine deadline the sweep itself wrote into the past — through the guard, since that is a statement it admits. + /// + /// Winds the clock past the quarantine the sweep itself wrote: the deadline lands in the past, and the row looks + /// settled two days ago rather than a moment ago, so the deferral gate is open again. Both are time travel — the + /// guard rightly refuses to move a modification time backwards — so it runs with the trigger suspended, exactly + /// like . That the reaper's OWN quarantine write is admitted by the guard is not assumed + /// here: it is what put a value in retain_until for this helper to move. + /// private async Task ElapseQuarantineAsync(Guid streamId) { + // GREATEST so the modification time never falls below the clocks ck_agent_run_log_stream_time orders it + // against: a stream whose capture was aged into the past is already older than the recheck interval anyway. + const string sql = "UPDATE agent_run_log_stream SET retain_until = now() - interval '1 second', revision = revision + 1, " + + "last_modified_at = GREATEST(created_at, capture_finalized_at, completed_at, last_modified_at - interval '2 days') WHERE id = {0}"; using var scope = _fixture.BeginScope(); - var updated = await scope.Resolve().Database.ExecuteSqlRawAsync( - "UPDATE agent_run_log_stream SET retain_until = now() - interval '1 second', revision = revision + 1, last_modified_at = now() WHERE id = {0}", [streamId]); + var db = scope.Resolve(); + await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream DISABLE TRIGGER agent_run_log_stream_enforce_invariants"); - updated.ShouldBe(1); + try { (await db.Database.ExecuteSqlRawAsync(sql, [streamId])).ShouldBe(1); } + finally { await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream ENABLE TRIGGER agent_run_log_stream_enforce_invariants"); } + } + + /// Drives one stream all the way through the plane — quarantine, elapse, collect — for the tests whose subject is what happens AFTER a purge. + private async Task PurgeAsync(World world, Guid streamId) + { + await AgeAsync(streamId, TimeSpan.FromDays(31)); + var quarantine = await SweepAsync(); + await ElapseQuarantineAsync(streamId); + var collection = await SweepAsync(); + + (await StreamAsync(streamId)).PurgedAt.ShouldNotBeNull( + $"the premise of this test is a stream the plane actually reclaimed; quarantine pass {Describe(quarantine)}, collection pass {Describe(collection)}"); + (await UnpurgedLocationsAsync(world, streamId)).ShouldBe(0); } /// - /// The real reaper over the real cursor, with only the CAS purge verdict swapped for the one a destination that - /// will not give the bytes up produces. Everything that decides whether the tombstone is written is production - /// code; the refusal is the single injected fact. + /// Takes every location behind this stream's objects out of the purge lifecycle without purging it. Deleted + /// is the state that says so: 0127's trigger makes it terminal, so unlike Missing — which a purge can still + /// drive to Purged — these bytes left by a path this plane will never drive. Its guards are suspended for + /// the one statement, exactly as the stream's are for time travel. /// - private async Task SweepAgainstARefusingDestinationAsync() + private async Task MarkLocationsDeletedAsync(World world, Guid streamId) { using var scope = _fixture.BeginScope(); - var options = scope.Resolve>(); - var cursor = new LogStreamRetentionCursor(options, new RefusingPurgeCoordinator(scope.Resolve()), NullLogger.Instance); + var db = scope.Resolve(); + var objects = await ObjectsOfAsync(world, streamId); + await db.Database.ExecuteSqlRawAsync("ALTER TABLE artifact_location DISABLE TRIGGER USER"); - return await new DurableRetentionReaper(options, [cursor], NullLogger.Instance).SweepAsync(CancellationToken.None); + try + { + await db.ArtifactLocation.Where(location => location.TeamId == world.TeamId && objects.Contains(location.ArtifactObjectId)) + .ExecuteUpdateAsync(set => set + .SetProperty(location => location.State, ArtifactLocationState.Deleted) + .SetProperty(location => location.Revision, location => location.Revision + 1)); + } + finally + { + await db.Database.ExecuteSqlRawAsync("ALTER TABLE artifact_location ENABLE TRIGGER USER"); + } } - /// A destination that answers every delete with a refusal and no effect — the shape a revoked key or a read-only bucket produces. - private sealed class RefusingPurgeCoordinator : IArtifactCasPurgeCoordinator + /// An intent that has not finished and tries to name the object it is moving — the row the schema refuses. + private static async Task InFlightIntentAsync(CodeSpaceDbContext db, World world, Guid objectId) { - private readonly IArtifactCasPurgeCoordinator _inner; - - public RefusingPurgeCoordinator(IArtifactCasPurgeCoordinator inner) { _inner = inner; } - - public Task PurgeAsync(ArtifactCasPurgeRequest request, CancellationToken cancellationToken) => - Task.FromResult(new ArtifactCasPurgeResult.Rejected(new ArtifactCasProblem(ArtifactCasProblemCode.ProviderUnavailable, true))); + var revisionId = await db.StorageProfileRevision.AsNoTracking() + .Where(row => row.StorageProfileId == world.StorageProfileId && row.Revision == 1).Select(row => row.Id).SingleAsync(); - public Task ClaimAsync(ArtifactCasPurgeRequest request, CancellationToken cancellationToken) => _inner.ClaimAsync(request, cancellationToken); - public Task DeleteAsync(ArtifactCasPurgeClaim claim, CancellationToken cancellationToken) => _inner.DeleteAsync(claim, cancellationToken); - public Task ReleaseAsync(ArtifactCasPurgeClaim claim, ArtifactCasReleaseEvidence evidence, CancellationToken cancellationToken) => _inner.ReleaseAsync(claim, evidence, cancellationToken); - public Task AbandonAsync(ArtifactCasPurgeClaim claim, CancellationToken cancellationToken) => _inner.AbandonAsync(claim, cancellationToken); + return new ArtifactTransferIntent + { + Id = Guid.NewGuid(), TeamId = world.TeamId, ArtifactObjectId = objectId, StorageProfileRevisionId = revisionId, + IdempotencyKey = $"retention-test/{objectId:N}", ExpectedDigest = System.Security.Cryptography.SHA256.HashData([7]), + ExpectedSizeBytes = 1, TargetLocator = "local-rwx", TargetObjectKey = $"transfer/{objectId:N}", + State = ArtifactTransferState.Intended, Revision = 1, CreatedDate = DateTimeOffset.UtcNow, CreatedBy = world.ActorId, + LastModifiedDate = DateTimeOffset.UtcNow, LastModifiedBy = world.ActorId, + }; } // ── What a sealed result cites ───────────────────────────────────────────────────────────────────────────────── @@ -398,15 +743,86 @@ await store.SealAsync(new PairedQualificationSealRequest CapabilityVerdictCells = 1, ObservedModelCells = 1, EvaluatorHealth = 1, ObservedModels = ["model"], }; + private static string Identity(PairedQualificationResultPin pin) => $"{pin.Kind}:{pin.Target}"; + + private async Task ForgetPinsAsync(Guid resultId) + { + using var scope = _fixture.BeginScope(); + await scope.Resolve().PairedQualificationResultPin.Where(pin => pin.ResultId == resultId).ExecuteDeleteAsync(); + } + + /// + /// Runs migration 0235's own backfill statements, read from the script that ships with the build rather than + /// copied here — a mirror would keep passing after the migration changed underneath it. + /// + private async Task RunBackfillAsync() + { + var script = await File.ReadAllTextAsync(Path.Combine(AppContext.BaseDirectory, "Persistence", "DbUpFiles", "0235_durable_retention_pins.sql")); + var statements = script.Split(';').Select(value => value.Trim()).Where(value => value.Contains("INSERT INTO paired_qualification_result_pin", StringComparison.Ordinal)).ToList(); + + statements.Count.ShouldBe(4, "0235 backfills four closures — runs, then their streams, receipts and offloaded payloads; if that count changed, so did what a pre-existing result protects"); + using var scope = _fixture.BeginScope(); + + foreach (var statement in statements) + await scope.Resolve().Database.ExecuteSqlRawAsync(statement[statement.IndexOf("INSERT INTO", StringComparison.Ordinal)..]); + } + // ── Reads ────────────────────────────────────────────────────────────────────────────────────────────────────── - private async Task SweepAsync() + private static string Describe(DurableRetentionSweepSummary summary) => + $"[claimed {summary.Claimed}, quarantined {summary.Quarantined}, collected {summary.Collected}, referenced {summary.Referenced}, kept {summary.Kept}]"; + + private async Task SweepAsync() { using var scope = _fixture.BeginScope(); return await scope.Resolve().SweepAsync(CancellationToken.None); } + /// + /// The real reaper over the real cursor, with only the CAS purge verdict swapped for the one a destination that + /// will not give the bytes up produces. Everything that decides whether the tombstone is written is production + /// code; the refusal is the single injected fact. + /// + private async Task SweepAgainstARefusingDestinationAsync() + { + using var scope = _fixture.BeginScope(); + var options = scope.Resolve>(); + var cursor = new LogStreamRetentionCursor(options, new RefusingPurgeCoordinator(), NullLogger.Instance); + + return await new DurableRetentionReaper(options, [cursor], NullLogger.Instance).SweepAsync(CancellationToken.None); + } + + /// A destination that answers every delete with a refusal and no effect — the shape a revoked key or a read-only bucket produces. + private sealed class RefusingPurgeCoordinator : IArtifactCasPurgeCoordinator + { + public Task PurgeAsync(ArtifactCasPurgeRequest request, CancellationToken cancellationToken) => + Task.FromResult(new ArtifactCasPurgeResult.Rejected(new ArtifactCasProblem(ArtifactCasProblemCode.ProviderUnavailable, true))); + + public Task ClaimAsync(ArtifactCasPurgeRequest request, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task DeleteAsync(ArtifactCasPurgeClaim claim, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ReleaseAsync(ArtifactCasPurgeClaim claim, ArtifactCasReleaseEvidence evidence, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task AbandonAsync(ArtifactCasPurgeClaim claim, CancellationToken cancellationToken) => throw new NotSupportedException(); + } + + private LogStreamRetentionCursor Cursor() + { + using var scope = _fixture.BeginScope(); + + return new LogStreamRetentionCursor(scope.Resolve>(), scope.Resolve(), NullLogger.Instance); + } + + /// What the production claim query returns for the window the loop would build right now. + private async Task> CandidatesAsync(int limit) + { + var rule = DurableRetentionPolicy.LogStream; + var now = DateTimeOffset.UtcNow; + + return await Cursor().ClaimAsync(new DurableRetentionSweepWindow(now, now - rule.MinimumAge, now - rule.RecheckInterval), limit, CancellationToken.None); + } + + private async Task> ClaimAsync(int limit) => (await CandidatesAsync(limit)).Select(candidate => candidate.Id).ToList(); + private async Task StreamAsync(Guid streamId) { using var scope = _fixture.BeginScope(); @@ -432,6 +848,15 @@ private async Task> InsertingTransactionsAsync(Guid result + "UNION SELECT DISTINCT xmin::text FROM paired_qualification_result_pin WHERE result_id = {0}", resultId).ToListAsync(); } + private async Task> ObjectsOfAsync(World world, Guid streamId) + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().AgentRunLogSegment.AsNoTracking() + .Where(segment => segment.TeamId == world.TeamId && segment.StreamId == streamId) + .Select(segment => segment.ArtifactObjectId).Distinct().OrderBy(id => id).ToListAsync(); + } + private async Task UnpurgedLocationsAsync(World world, Guid streamId) { using var scope = _fixture.BeginScope(); @@ -441,29 +866,38 @@ private async Task UnpurgedLocationsAsync(World world, Guid streamId) .Where(location => location.TeamId == world.TeamId && location.State != ArtifactLocationState.Purged && location.State != ArtifactLocationState.Deleted) .Join(db.AgentRunLogSegment.AsNoTracking().Where(segment => segment.StreamId == streamId), location => location.ArtifactObjectId, segment => segment.ArtifactObjectId, (location, _) => location.Id) + .Distinct() .CountAsync(); } - /// Whether the stream's bytes still come back through the production read path — the only reading of "the bytes are there" that matters. - private async Task BytesReadableAsync(World world, Guid streamId) + private async Task MetadataAsync(World world, Guid streamId) + { + using var scope = _fixture.BeginScope(); + + return (await Logs(scope).GetMetadataAsync(world.TeamId, streamId, CancellationToken.None)).ShouldBeOfType().Metadata; + } + + private async Task ReadAsync(World world, Guid streamId) { using var scope = _fixture.BeginScope(); var stream = await StreamAsync(streamId); - var read = await Logs(scope).ReadRangeAsync(new AgentRunLogRangeRequest(world.TeamId, streamId, 0, checked((int)stream.TotalBytes)), CancellationToken.None); - return read is AgentRunLogRangeResult.Available; + return await Logs(scope).ReadRangeAsync(new AgentRunLogRangeRequest(world.TeamId, streamId, 0, checked((int)stream.TotalBytes)), CancellationToken.None); } + /// Whether the stream's bytes still come back through the production read path — the only reading of "the bytes are there" that matters. + private async Task BytesReadableAsync(World world, Guid streamId) => await ReadAsync(world, streamId) is AgentRunLogRangeResult.Available; + private static RoomProjector.AgentLogRow Row(AgentRunLogStream stream) => new(stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null, stream.PurgedAt != null); private static AgentRunLogService Logs(ILifetimeScope scope) => new(scope.Resolve>(), scope.Resolve(), TimeProvider.System); - private static AgentRunLogOpenRequest Open(World world, Guid session) => new() + private static AgentRunLogOpenRequest Open(World world, Guid session, string kind = AgentRunLogKinds.StandardOutput) => new() { TeamId = world.TeamId, AgentRunId = world.AgentRunId, WorkerFenceEpoch = Fence, CaptureSessionId = session, - StreamKind = AgentRunLogKinds.StandardOutput, ContentType = AgentRunLogRepresentations.PlainTextContentType, + StreamKind = kind, ContentType = AgentRunLogRepresentations.PlainTextContentType, ContentEncoding = AgentRunLogRepresentations.Utf8ContentEncoding, CaptureSource = "retention-test/v1", }; @@ -535,13 +969,59 @@ private string NewRoot() return root; } - public void Dispose() + public Task InitializeAsync() => Task.CompletedTask; + + /// + /// Reclaims every placement this class staged, through the same purge lifecycle the production plane uses, BEFORE + /// the destinations it wrote them to are removed. + /// + /// This is not tidiness. The artifact-location verifier is a bounded deployment-wide sweep — the hundred + /// least-recently-verified placements across every team, on a database this collection shares — and its own tests + /// assert that it reached THEIR rows. Live placements left behind a deleted local root change both what that batch + /// contains and which destinations answer in it, so a class that stages placements owns them all the way to the + /// end. + /// + public async Task DisposeAsync() { + foreach (var staged in _staged) await ReclaimAsync(staged); + foreach (var root in _roots) { try { if (Directory.Exists(root)) Directory.Delete(root, recursive: true); } catch { /* best-effort */ } } } + private async Task ReclaimAsync(StagedStream staged) + { + try + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + var purge = scope.Resolve(); + var objects = db.AgentRunLogSegment.AsNoTracking() + .Where(segment => segment.TeamId == staged.TeamId && segment.StreamId == staged.StreamId) + .Select(segment => segment.ArtifactObjectId).Distinct(); + + // Per LOCATION, not per object: identical captures deduplicate to one object with a placement per write, + // and a claim that does not name which placement refuses a replicated object outright — which is the + // production behaviour, and the reason this cleanup has to be explicit about it. + var placements = await db.ArtifactLocation.AsNoTracking() + .Where(location => location.TeamId == staged.TeamId && objects.Contains(location.ArtifactObjectId) + && location.State != ArtifactLocationState.Purged && location.State != ArtifactLocationState.Deleted) + .Select(location => new { location.Id, location.ArtifactObjectId }) + .ToListAsync(); + + foreach (var placement in placements) + await purge.PurgeAsync(new ArtifactCasPurgeRequest + { + TeamId = staged.TeamId, ArtifactObjectId = placement.ArtifactObjectId, + ArtifactLocationId = placement.Id, ActorId = Messages.Constants.SystemUsers.SeederId, + }, CancellationToken.None); + } + catch { /* best-effort: a test that already purged, or left a refusing destination, has nothing more to give back */ } + } + + private sealed record StagedStream(Guid TeamId, Guid StreamId); + private sealed record World(Guid TeamId, Guid ActorId, Guid StorageProfileId, Guid AgentRunId, Guid ControlModelRowId, Guid CandidateModelRowId); } diff --git a/backend/tests/CodeSpace.UnitTests/Persistence/AgentRunLogSchemaTests.cs b/backend/tests/CodeSpace.UnitTests/Persistence/AgentRunLogSchemaTests.cs index 48457f7be..e72c4f306 100644 --- a/backend/tests/CodeSpace.UnitTests/Persistence/AgentRunLogSchemaTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Persistence/AgentRunLogSchemaTests.cs @@ -28,6 +28,11 @@ public void Stream_is_a_tenant_bound_monotonic_head_for_one_open_versioned_log_k "AgentRunId", "CaptureFinalizedAt", "CaptureSessionId", "CaptureSource", "CaptureSourceBaseOffsetBytes", "CompletedAt", "ContentDigest", "ContentDigestAlgorithm", "ContentEncoding", "ContentType", "CreatedAt", "ErrorCode", "ErrorMessage", "ExpiresAt", "Id", "LastModifiedAt", "ManifestDigest", "NextOffsetBytes", "NextSegmentOrdinal", "RemoteStallCode", "RemoteStallSince", "Retention", "Revision", "SchemaVersion", "SegmentCount", "State", + // The retention plane's two columns (migration 0235), and the only part of this row a process that is not + // the capturing worker may ever write: retain_until is the instant the bytes become reclaimable, purged_at + // the tombstone the head row keeps after they are gone. The guard admits them on a TERMINAL stream only, + // alone, and never in the same statement as anything else — so a purge can never pass for a capture verdict. + "PurgedAt", "RetainUntil", "SourceOffsetBytes", "StreamKind", "TeamId", "TotalBytes", "WorkerFenceEpoch", "Xmin", }.Order()); entity.FindProperty(nameof(AgentRunLogStream.State))!.GetMaxLength().ShouldBe(24); diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs index 33a2dd854..4b48a1661 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs @@ -18,26 +18,26 @@ public sealed class DurableRetentionPolicyTests private static readonly DateTimeOffset Now = new(2026, 9, 18, 12, 0, 0, TimeSpan.Zero); private static readonly DurableRetentionRule Rule = DurableRetentionPolicy.LogStream; - [Theory] - [InlineData(DurableRecordClass.LogStream, 30)] - [InlineData(DurableRecordClass.CleanupReceipt, 30)] - [InlineData(DurableRecordClass.CaptureGap, 30)] - [InlineData(DurableRecordClass.QualificationEvidence, 180)] - [InlineData(DurableRecordClass.BudgetReservation, 90)] - [InlineData(DurableRecordClass.TransferIntent, 7)] - public void The_committed_rule_table_is_pinned_to_its_literal_windows(DurableRecordClass value, int minimumAgeDays) + [Fact] + public void The_committed_rule_table_is_pinned_to_its_literal_windows() { - var rule = DurableRetentionPolicy.For(value).ShouldNotBeNull(); + var rule = DurableRetentionPolicy.For(DurableRecordClass.LogStream).ShouldNotBeNull(); - rule.MinimumAge.ShouldBe(TimeSpan.FromDays(minimumAgeDays)); - rule.QuarantineWindow.ShouldBe(TimeSpan.FromHours(24), "every class waits a second, independent day after the first uncited observation"); + rule.MinimumAge.ShouldBe(TimeSpan.FromDays(30), "a log stream is not a candidate until a month after its capture settled"); + rule.QuarantineWindow.ShouldBe(TimeSpan.FromHours(24), "a second, independent day passes after the first uncited observation"); + rule.RecheckInterval.ShouldBe(TimeSpan.FromHours(24), "a stream that was looked at and kept is left alone this long, so one unreclaimable row cannot own a batch slot"); } [Fact] public void Every_declared_class_has_a_rule_and_the_table_declares_nothing_else() { + // The table advertises what is actually reclaimed. A class listed here without a cursor would read as a + // promise the system does not keep; a cursor whose class is missing claims nothing at all. Either way the + // right time to notice is here. DurableRetentionPolicy.Rules.Keys.Order().ShouldBe(Enum.GetValues().Order(), customMessage: "a class with no rule is never claimed, so adding one to the enum without a rule silently disables its plane"); + Enum.GetValues().ShouldBe([DurableRecordClass.LogStream], + customMessage: "a class belongs here only together with the cursor that sweeps it — add both in one change, never the rule first"); } [Fact] @@ -49,6 +49,29 @@ public void A_class_the_policy_does_not_register_has_no_rule_and_therefore_keeps Decide(null, Now.AddDays(-400), null, DurableReferenceVerdict.Unreferenced).Action.ShouldBe(DurableRetentionAction.Indeterminate); } + [Fact] + public void A_citation_clears_the_quarantine_it_contradicts() + { + // Mutation: carry the old retain_until through a Referenced settlement. A stream cited for a year would then + // be collectable the instant its pin went away, with a quarantine "window" that elapsed while something still + // pointed at it — a second wait that never actually waited for anything. + var decision = Decide(Rule, Now.AddDays(-400), Now.AddDays(-100), DurableReferenceVerdict.Referenced); + + decision.RetainUntil.ShouldBeNull(); + } + + [Fact] + public void An_unanswered_question_keeps_the_quarantine_it_never_contradicted() + { + // The opposite of the test above, and the reason the two are separate: an unreadable citation site says + // nothing about the earlier uncited observation, so clearing the marker would restart a wait that was already + // most of the way through. + var quarantinedAt = Now.AddHours(-1); + var decision = Decide(Rule, Now.AddDays(-400), quarantinedAt, DurableReferenceVerdict.Indeterminate); + + decision.RetainUntil.ShouldBe(quarantinedAt); + } + [Fact] public void Pinned_row_is_never_collected() { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs new file mode 100644 index 000000000..410e6158f --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs @@ -0,0 +1,77 @@ +using CodeSpace.Core.Persistence.Db; +using CodeSpace.Core.Persistence.Entities; +using CodeSpace.Core.Services.Workflows.Retention.Cursors; +using Microsoft.EntityFrameworkCore; +using Shouldly; + +namespace CodeSpace.UnitTests.Workflows.Retention; + +/// +/// The enumeration of everything that can still name a log stream's CAS object. It is pinned for the same reason +/// ArtifactReferenceOracle.ReferenceSites is: no foreign key points at an artifact_object from most of +/// these places, so the list IS the completeness argument, and a citer missing from it makes the cursor answer +/// "unreferenced" about bytes something still reaches. +/// +/// The drift detector below is the half that cannot be forgotten: it reads the EF model rather than this list, +/// so a future column that names an artifact object reds here even though nobody thought about retention while adding +/// it. +/// +[Trait("Category", "Unit")] +public sealed class LogStreamCitationSitesTests +{ + private const string UnreachableDatabase = "Host=127.0.0.1;Port=1;Database=unused;Username=unused;Password=unused"; + + [Fact] + public void Every_citer_of_a_log_streams_bytes_is_pinned() + { + LogStreamRetentionCursor.CitationSites.ShouldBe(new[] + { + // The qualification pin: a sealed result still cites this stream. + ("paired_qualification_result_pin", "pinned_id"), + + // Another stream's segment. The CAS is content-addressed, so two captures of identical output — the + // ordinary case for short or empty streams — are ONE object under two names. + ("agent_run_log_segment", "artifact_object_id"), + + // An artifact stored as the same routed object. + ("workflow_artifact", "cas_artifact_object_id"), + }, ignoreOrder: true); + } + + [Fact] + public void A_mapped_column_that_names_an_artifact_object_and_is_not_probed_fails_this_test() + { + using var db = BuildContext(); + var probed = LogStreamRetentionCursor.CitationSites.Select(site => $"{site.Table}.{site.Column}").ToHashSet(StringComparer.Ordinal); + + var mapped = db.Model.GetEntityTypes() + .SelectMany(entity => entity.GetProperties().Select(property => (Table: entity.GetTableName(), Column: property.GetColumnName()))) + .Where(column => column.Table is not null && column.Column is not null && IsObjectSoftLink(column.Table!, column.Column!)) + .Select(column => $"{column.Table}.{column.Column}") + .Distinct(StringComparer.Ordinal) + .Order(StringComparer.Ordinal) + .ToList(); + + mapped.Where(column => !probed.Contains(column)).ShouldBeEmpty( + "a column that names an artifact_object which the log-stream cursor does not probe would let it purge bytes that row still reaches — " + + $"add it to {nameof(LogStreamRetentionCursor)}.{nameof(LogStreamRetentionCursor.CitationSites)} and to the probe beside it"); + } + + /// + /// A column names a CAS object when it ends in artifact_object_id. Three tables are excluded, each for a + /// stated reason rather than because it was inconvenient: + /// + /// artifact_object and artifact_location are the object's own identity and its placements, + /// not a second holder of its bytes — the cursor reads both directly, and a purge is precisely what advances a + /// location's state. artifact_transfer_intent is excluded because a saga in flight CANNOT name an object: + /// ck_artifact_transfer_intent_outcome requires the column to be null on every non-terminal row, so every + /// row that names one is a finished transfer, and a transfer is what wrote each log segment in the first place. + /// An integration test asserts that refusal rather than trusting this comment. + /// + private static bool IsObjectSoftLink(string table, string column) => + column.EndsWith("artifact_object_id", StringComparison.Ordinal) + && table is not ("artifact_object" or "artifact_location" or "artifact_cas_purge_claim" or "artifact_transfer_intent"); + + private static CodeSpaceDbContext BuildContext() => + new(new DbContextOptionsBuilder().UseNpgsql(UnreachableDatabase).UseSnakeCaseNamingConvention().Options); +} diff --git a/frontend/src/api/sessions.ts b/frontend/src/api/sessions.ts index 8fc9a107f..aac8b387f 100644 --- a/frontend/src/api/sessions.ts +++ b/frontend/src/api/sessions.ts @@ -337,7 +337,7 @@ export interface RoomArtifactVerification { } /// One agent's durable log-stream health, as the Room's projector folds it. -export type RoomAgentLogStatus = "Verified" | "Captured" | "Finalizing" | "Incomplete" | "Stalled"; +export type RoomAgentLogStatus = "Verified" | "Captured" | "Finalizing" | "Incomplete" | "Stalled" | "Purged"; /// What the sandbox actually did to one producer. `Unknown` is a real absence (nothing recorded it), never a /// confinement to render as safety — the renderer must say "posture unknown" rather than show a confined glyph. diff --git a/frontend/src/components/sessions/SessionRoomView.tsx b/frontend/src/components/sessions/SessionRoomView.tsx index d881e6c73..6b2e42ca7 100644 --- a/frontend/src/components/sessions/SessionRoomView.tsx +++ b/frontend/src/components/sessions/SessionRoomView.tsx @@ -2005,6 +2005,7 @@ const PRODUCER_LOG_LABEL: Record = { Finalizing: "logs finalizing", Incomplete: "logs incomplete", Stalled: "logs held; storage unavailable", + Purged: "logs purged; retention window elapsed", }; /** From f489b2e76c3f9649e5c2b75161a32474e1b9db30 Mon Sep 17 00:00:00 2001 From: "Mars.P" Date: Sat, 19 Sep 2026 14:01:00 +0800 Subject: [PATCH 4/4] Purge a log stream by placement, and reach past the oldest stream per team MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three things the first rework got wrong, each of which quietly kept bytes nobody wanted. Deduplication is per WRITE, not per stream: two byte-identical segments are one content-addressed object with a placement each, and because the object key is chosen per write both can sit at the same destination. The drain refused any object with more than one placement, so a capture that repeated itself was unreclaimable for good — and the suite could not see it, because every test wrote distinct bytes per segment. The drain now walks placements and names each one in its purge request, which is what the coordinator asks for when an object has more than one. The refusal it replaced was also misreported: "holds bytes at N destinations" was one destination, N keys. The per-tenant fair claim returns one record per team, so claiming once capped the sweep at one stream per team per tick — about a dozen a day against the two a single run writes. The loop asks again until the batch is full, skipping ids already settled this tick so a drain that is still making progress cannot spin. And the log reader's own frontend rejected the new Purged verdict as a malformed response, which is worse than the wrong-but-well-formed answer it replaced: a 410 an operator could read became "invalid log response". --- .../Cursors/LogStreamRetentionCursor.cs | 96 +++++++++---------- .../Retention/DurableRetentionReaper.cs | 24 ++++- .../DurableRetentionReaperFlowTests.cs | 61 ++++++++++-- .../Retention/DurableRetentionPolicyTests.cs | 15 +++ .../Retention/LogStreamCitationSitesTests.cs | 56 ++++++++++- frontend/src/api/agentRunLogsApi.test.ts | 12 +++ frontend/src/api/agents.ts | 7 +- .../src/components/workflows/AgentRunLogs.tsx | 1 + 8 files changed, 204 insertions(+), 68 deletions(-) diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs index f115884f0..510d36c9e 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs @@ -28,23 +28,26 @@ namespace CodeSpace.Core.Services.Workflows.Retention.Cursors; /// the question could not be asked: every failure answers , never /// "unreferenced". /// -/// What this cursor cannot reclaim, and says so. A REPLICATED object — one whose bytes sit at more than -/// one destination — cannot be purged at all: IArtifactCasPurgeCoordinator refuses it before touching any -/// location, because deleting every replica needs an object-level claim the schema does not yet represent. This -/// cursor names that case from the locations it has already read rather than from the refusal, since the coordinator -/// rejects it with the same code as "no location at all" and a caller that could not tell them apart would retry a -/// call whose answer can never change. Such a stream is kept and logged under its own message; a deployment that -/// replicates its log storage reclaims nothing here until that claim exists. +/// Placements, not objects. Two byte-identical writes deduplicate to ONE content-addressed object with a +/// placement per write — and, because the object key is chosen per write, both can sit at the SAME destination. The +/// drain therefore walks placements and NAMES each one in its purge request: an unnamed claim refuses an object with +/// more than one placement outright (the coordinator will not guess which), which would make a stream that captured +/// the same bytes twice permanently unreclaimable. +/// +/// A placement the destination will not give up — a Corrupt one, which is claimable but never +/// deletable, or a destination that is refusing — is a keep, logged with the refusal's own code on every sweep and +/// deferred for the class's recheck interval. Deliberately not a terminal state: this plane has nowhere to record +/// one, and a refusal that becomes silent is how bytes stop being accounted for. /// public sealed class LogStreamRetentionCursor : IDurableRetentionCursor, IScopedDependency { /// - /// Objects purged per stream per sweep. A 4 GiB capture is thousands of segments, and a sweep that tried to drain + /// Placements purged per stream per sweep. A 4 GiB capture is thousands of segments, and a sweep that tried to drain /// one in a single pass would hold a worker for minutes on provider I/O. A partial drain deliberately writes /// NOTHING, so the stream is re-claimed on the very next tick and continues where it stopped — unlike a refusal, /// which defers it for the class's recheck interval. /// - internal const int MaxObjectsPerSweep = 64; + internal const int MaxPlacementsPerSweep = 64; /// /// Every place a log stream's CAS object can still be named, as (what, why). This list IS the safety of the @@ -190,17 +193,16 @@ private async Task DrainAsync(DurableRetentionCandidate candidate, if (!await objects.AnyAsync(cancellationToken).ConfigureAwait(false)) return NothingCaptured(candidate); - var reclaimable = await ReclaimableAsync(db, candidate, objects, cancellationToken).ConfigureAwait(false); + var placements = await ReclaimableAsync(db, candidate, objects, cancellationToken).ConfigureAwait(false); - if (reclaimable.Count == 0) return await DrainedOrLostAsync(db, candidate, objects, cancellationToken).ConfigureAwait(false); - if (reclaimable.FirstOrDefault(row => row.LiveLocations > 1) is { } replicated) return Replicated(candidate, replicated); + if (placements.Count == 0) return await DrainedOrLostAsync(db, candidate, objects, cancellationToken).ConfigureAwait(false); - foreach (var row in reclaimable.Take(MaxObjectsPerSweep)) + foreach (var placement in placements.Take(MaxPlacementsPerSweep)) { - if (!await PurgeObjectAsync(candidate, row.ObjectId, cancellationToken).ConfigureAwait(false)) return DrainOutcome.Refused; + if (!await PurgePlacementAsync(candidate, placement, cancellationToken).ConfigureAwait(false)) return DrainOutcome.Refused; } - return reclaimable.Count > MaxObjectsPerSweep ? DrainOutcome.Progressing : DrainOutcome.Drained; + return placements.Count > MaxPlacementsPerSweep ? DrainOutcome.Progressing : DrainOutcome.Drained; } /// The objects this stream's segments name — the entry point for every question the drain asks. @@ -211,26 +213,24 @@ private static IQueryable Objects(CodeSpaceDbContext db, DurableRetentionC .Distinct(); /// - /// The next objects whose bytes are still at a destination, each with how many destinations hold it, and one row - /// over the batch so the caller can see whether another sweep is needed. Read inside the collecting pass rather - /// than carried from the decision, so a resumed drain converges instead of re-asking about objects an earlier - /// pass already took. + /// The next placements whose bytes are still at a destination, one row over the batch so the caller can see + /// whether another sweep is needed. Read inside the collecting pass rather than carried from the decision, so a + /// resumed drain converges instead of re-asking about placements an earlier pass already took. /// - private static async Task> ReclaimableAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, IQueryable objects, CancellationToken cancellationToken) + private static async Task> ReclaimableAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, IQueryable objects, CancellationToken cancellationToken) { - // Projected through an anonymous type rather than straight into the record: a positional constructor inside a - // GROUP BY projection is not translatable, and the exception it raises is swallowed as "could not be - // evaluated" — which reads exactly like a fail-closed keep and hides a cursor that reclaims nothing. + // Projected through an anonymous type rather than straight into the record: a positional constructor is not + // always translatable, and the exception it raises is swallowed as "could not be evaluated" — which reads + // exactly like a fail-closed keep and hides a cursor that reclaims nothing. var rows = await db.ArtifactLocation.AsNoTracking() .Where(location => location.TeamId == candidate.TeamId && objects.Contains(location.ArtifactObjectId) && location.State != ArtifactLocationState.Purged && location.State != ArtifactLocationState.Deleted) - .GroupBy(location => location.ArtifactObjectId) - .Select(group => new { ObjectId = group.Key, LiveLocations = group.Count() }) - .OrderBy(row => row.ObjectId) - .Take(MaxObjectsPerSweep + 1) + .OrderBy(location => location.ArtifactObjectId).ThenBy(location => location.Id) + .Select(location => new { LocationId = location.Id, location.ArtifactObjectId }) + .Take(MaxPlacementsPerSweep + 1) .ToListAsync(cancellationToken).ConfigureAwait(false); - return rows.Select(row => new SegmentObject(row.ObjectId, row.LiveLocations)).ToArray(); + return rows.Select(row => new SegmentPlacement(row.LocationId, row.ArtifactObjectId)).ToArray(); } /// @@ -247,19 +247,6 @@ private async Task DrainedOrLostAsync(CodeSpaceDbContext db, Durab return unaccounted == 0 ? DrainOutcome.Drained : BytesAlreadyGone(candidate, unaccounted); } - /// - /// Named here rather than read off a refusal code, because the coordinator cannot tell this refusal from "no - /// location at all" — both reject with ArtifactMissing, and a cursor that guessed between them would keep - /// retrying a call whose answer can never change. The locations this drain already read say it exactly. - /// - private DrainOutcome Replicated(DurableRetentionCandidate candidate, SegmentObject replicated) - { - _logger.LogWarning("Log stream {StreamId}: object {ObjectId} holds bytes at {LocationCount} destinations, which this plane cannot reclaim — deleting every replica needs an object-level purge claim that does not exist yet, so the stream is kept", - candidate.Id, replicated.ObjectId, replicated.LiveLocations); - - return DrainOutcome.Replicated; - } - private DrainOutcome NothingCaptured(DurableRetentionCandidate candidate) { _logger.LogInformation("Log stream {StreamId}: no segment ever landed, so there are no bytes to reclaim and no purge to record", candidate.Id); @@ -275,9 +262,18 @@ private DrainOutcome BytesAlreadyGone(DurableRetentionCandidate candidate, int u return DrainOutcome.BytesAlreadyGone; } - private async Task PurgeObjectAsync(DurableRetentionCandidate candidate, Guid objectId, CancellationToken cancellationToken) + /// + /// One placement, NAMED. Leaving ArtifactLocationId null means "the only one", which an object with more + /// than one placement cannot answer — the coordinator refuses rather than guessing, so an unnamed claim would + /// make every deduplicated capture unreclaimable for good. + /// + private async Task PurgePlacementAsync(DurableRetentionCandidate candidate, SegmentPlacement placement, CancellationToken cancellationToken) { - var request = new ArtifactCasPurgeRequest { TeamId = candidate.TeamId, ArtifactObjectId = objectId, ActorId = SystemUsers.SeederId }; + var request = new ArtifactCasPurgeRequest + { + TeamId = candidate.TeamId, ArtifactObjectId = placement.ObjectId, + ArtifactLocationId = placement.LocationId, ActorId = SystemUsers.SeederId, + }; var outcome = await _purge.PurgeAsync(request, cancellationToken).ConfigureAwait(false); switch (outcome) @@ -285,10 +281,11 @@ private async Task PurgeObjectAsync(DurableRetentionCandidate candidate, G case ArtifactCasPurgeResult.Purged: return true; case ArtifactCasPurgeResult.Rejected rejected: - _logger.LogWarning("Log stream {StreamId}: object {ObjectId} was not reclaimed ('{Problem}'); the stream keeps its bytes and its head row", candidate.Id, objectId, rejected.Problem.Code); + _logger.LogWarning("Log stream {StreamId}: placement {LocationId} of object {ObjectId} was not reclaimed ('{Problem}'); the stream keeps its bytes and its head row", + candidate.Id, placement.LocationId, placement.ObjectId, rejected.Problem.Code); return false; default: - _logger.LogWarning("Log stream {StreamId}: object {ObjectId} returned an unrecognised purge outcome; the stream is kept", candidate.Id, objectId); + _logger.LogWarning("Log stream {StreamId}: placement {LocationId} returned an unrecognised purge outcome; the stream is kept", candidate.Id, placement.LocationId); return false; } } @@ -350,8 +347,8 @@ private async Task DatabaseClockAsync(CancellationToken cancella return await db.Database.SqlQueryRaw("SELECT clock_timestamp() AS \"Value\"").SingleAsync(cancellationToken).ConfigureAwait(false); } - /// One segment object as one read saw it: how many destinations still hold its bytes, and whether any location rests at Purged. - private sealed record SegmentObject(Guid ObjectId, int LiveLocations); + /// One placement still holding bytes for a segment object, as one read saw it. + private sealed record SegmentPlacement(Guid LocationId, Guid ObjectId); /// What one drain attempt established. Only permits a tombstone. private enum DrainOutcome @@ -362,12 +359,9 @@ private enum DrainOutcome /// Objects were removed and more remain. Write nothing: the next tick continues from here. Progressing, - /// The destination would not give the bytes up. Defer. + /// The destination would not give the bytes up — a refusing provider, or a Corrupt placement that is claimable but never deletable. Defer. Refused, - /// The bytes sit at more than one destination, which no claim in this schema can take. Defer, and say so. - Replicated, - /// Nothing was captured, or the bytes left by a path this plane did not drive. Never a purge. BytesAlreadyGone, } diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs index 17f64fb45..11e4010af 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs @@ -34,6 +34,9 @@ public sealed class DurableRetentionReaper : IDurableRetentionReaper /// Records claimed per cursor per sweep. The cadence is hourly and every rule is measured in days, so the ceiling changes only how promptly a backlog drains — never what is collected. private const int BatchSize = 200; + /// Records asked for per claim. A fair cursor returns the per-tenant head, so the batch above is reached by asking repeatedly rather than in one query, and a tenant with a long backlog never holds the whole batch. + private const int ClaimSize = 25; + private static readonly TimeSpan CandidateTimeout = TimeSpan.FromMinutes(2); private readonly DbContextOptions _dbOptions; @@ -70,11 +73,24 @@ private async Task SweepCursorAsync(IDurableRetentionCursor cursor, DateTimeOffs } var window = new DurableRetentionSweepWindow(now, now.Subtract(rule.MinimumAge), now.Subtract(rule.RecheckInterval)); - var candidates = await cursor.ClaimAsync(window, BatchSize, cancellationToken).ConfigureAwait(false); - counts.Claimed += candidates.Count; + var seen = new HashSet(); + + while (counts.Claimed < BatchSize) + { + var limit = Math.Min(ClaimSize, BatchSize - counts.Claimed); + var claimed = await cursor.ClaimAsync(window, limit, cancellationToken).ConfigureAwait(false); + // A cursor that claims fairly hands back one record per tenant, so the batch is reached by asking again — + // and the ids already settled this tick are excluded, because a settlement that writes nothing (a drain + // still making progress) leaves the record eligible and the loop would otherwise spin on it. + var candidates = claimed.Where(candidate => seen.Add(candidate.Id)).ToList(); - foreach (var candidate in candidates) - counts.Record(await SweepCandidateAsync(cursor, rule, candidate, now, cancellationToken).ConfigureAwait(false)); + if (candidates.Count == 0) break; + + counts.Claimed += candidates.Count; + + foreach (var candidate in candidates) + counts.Record(await SweepCandidateAsync(cursor, rule, candidate, now, cancellationToken).ConfigureAwait(false)); + } } /// One candidate, start to finish. Every exit that is not a completed settlement keeps the record. diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs index 1849834e1..f88fa0083 100644 --- a/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs @@ -159,6 +159,31 @@ public async Task A_stream_that_was_kept_is_not_re_claimed_until_its_recheck_int claimed.ShouldContain(collectable, "and the stream behind it, which no sweep has looked at, must be reachable"); } + /// + /// Throughput, which the per-tenant fairness of the claim query would otherwise cap at one record per tenant per + /// tick: with an hourly cadence and a two-visit collection that is about a dozen streams a team a day, while a + /// single run writes two. The loop asks again until the batch is full. Mutation: claim once and only the oldest + /// stream of the three is ever swept. + /// + [Fact] + public async Task One_sweep_reaches_every_eligible_stream_of_a_team_not_just_its_oldest() + { + var world = await SeedWorldAsync(); + var streams = new List(); + + foreach (var kind in new[] { AgentRunLogKinds.StandardOutput, AgentRunLogKinds.StandardError, AgentRunLogKinds.Transcript }) + { + var stream = await CaptureAsync(world, $"archive {kind}", kind); + await AgeAsync(stream, TimeSpan.FromDays(31)); + streams.Add(stream); + } + + await SweepAsync(); + + foreach (var stream in streams) + (await StreamAsync(stream)).RetainUntil.ShouldNotBeNull($"stream {stream} of the same team was never reached by the sweep"); + } + /// /// The atomicity claim, measured rather than asserted: xmin is the id of the transaction that inserted a /// row, so a result and its pins sharing one xmin IS "written in the same transaction". Mutation: move the @@ -247,7 +272,7 @@ public async Task A_refused_byte_purge_leaves_the_tombstone_unwritten_and_defers public async Task A_drain_too_big_for_one_sweep_writes_nothing_and_the_next_sweep_finishes_it() { var world = await SeedWorldAsync(); - var segments = LogStreamRetentionCursor.MaxObjectsPerSweep + 1; + var segments = LogStreamRetentionCursor.MaxPlacementsPerSweep + 1; var stream = await CaptureAsync(world, "first", AgentRunLogKinds.StandardOutput, segments); await AgeAsync(stream, TimeSpan.FromDays(31)); await SweepAsync(); @@ -259,7 +284,7 @@ public async Task A_drain_too_big_for_one_sweep_writes_nothing_and_the_next_swee var partly = await StreamAsync(stream); partly.PurgedAt.ShouldBeNull("the tombstone waits until every object is gone"); partly.LastModifiedAt.ShouldBe(beforeDrain.LastModifiedAt, "a drain that is progressing writes NOTHING, so the very next tick continues it"); - (await UnpurgedLocationsAsync(world, stream)).ShouldBe(segments - LogStreamRetentionCursor.MaxObjectsPerSweep, "exactly one bounded batch of objects went"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(segments - LogStreamRetentionCursor.MaxPlacementsPerSweep, "exactly one bounded batch of objects went"); await SweepAsync(); @@ -267,6 +292,29 @@ public async Task A_drain_too_big_for_one_sweep_writes_nothing_and_the_next_swee (await UnpurgedLocationsAsync(world, stream)).ShouldBe(0); } + /// + /// Deduplication INSIDE one stream. Two byte-identical segments are one content-addressed object with a placement + /// per write, both at the same destination, and nothing else names them — so the stream is collectable and the + /// drain has to take each placement by name. Mutation: purge by object id alone and the coordinator refuses to + /// guess which placement, so a capture that repeated itself would never be reclaimed at all. + /// + [Fact] + public async Task A_stream_that_captured_the_same_bytes_twice_is_still_collected() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "identical", segments: 2, distinctSegments: false); + (await ObjectsOfAsync(world, stream)).Count.ShouldBe(1, "the premise: the CAS stored ONE object for the two identical segments"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(2, "and ONE placement per write, which an unnamed purge claim refuses to choose between"); + + await AgeAsync(stream, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(stream); + await SweepAsync(); + + (await StreamAsync(stream)).PurgedAt.ShouldNotBeNull("nothing outside this stream names these bytes, so repeating itself must not make a capture unreclaimable"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(0, "every placement of the object goes, not just the one an unnamed claim would have found"); + } + /// /// Two streams that captured identical output are ONE content-addressed object under two names — the ordinary case /// for short or empty logs. Mutation: drop the sibling-segment probe and purging either one silently empties the @@ -508,7 +556,7 @@ public async Task The_log_archive_tables_refuse_every_deletion() // ── The world, and the durable streams inside it ─────────────────────────────────────────────────────────────── /// One real capture through the real service: an open, segments at a real local-rwx destination, a finalized source and a v3 completion. - private async Task CaptureAsync(World world, string text, string kind = AgentRunLogKinds.StandardOutput, int segments = 1) + private async Task CaptureAsync(World world, string text, string kind = AgentRunLogKinds.StandardOutput, int segments = 1, bool distinctSegments = true) { using var scope = _fixture.BeginScope(); var logs = Logs(scope); @@ -518,9 +566,10 @@ private async Task CaptureAsync(World world, string text, string kind = Ag for (var ordinal = 1; ordinal <= segments; ordinal++) { - // Distinct bytes per segment on purpose: the CAS is content-addressed, so repeating a payload would make - // two segments one object and the drain would have fewer objects to take than the test counted on. - var bytes = System.Text.Encoding.UTF8.GetBytes($"{text} #{ordinal:D4} {new string('x', 24)}"); + // Distinct bytes per segment by default: the CAS is content-addressed, so repeating a payload makes two + // segments ONE object with a placement each — which is its own case, and the caller says when it wants it. + var suffix = distinctSegments ? $" #{ordinal:D4}" : string.Empty; + var bytes = System.Text.Encoding.UTF8.GetBytes($"{text}{suffix} {new string('x', 24)}"); metadata = (await logs.AppendAsync(new AgentRunLogAppendRequest { TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = metadata.StreamId, WorkerFenceEpoch = Fence, diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs index 4b48a1661..6a93f6cf8 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs @@ -1,4 +1,5 @@ using CodeSpace.Core.Services.Workflows.Retention; +using CodeSpace.Messages.Constants; using CodeSpace.Messages.Retention; using Shouldly; @@ -40,6 +41,20 @@ public void Every_declared_class_has_a_rule_and_the_table_declares_nothing_else( customMessage: "a class belongs here only together with the cursor that sweeps it — add both in one change, never the rule first"); } + /// + /// Who a reclamation is attributed to. The seeder is the established identity for background work — Hangfire + /// workers, scheduled jobs and DbUp all write under it (SystemUsers.cs), and it holds a real seeded row, so + /// a purge receipt attributes to something that exists. A dedicated retention actor would be more specific, but it + /// would need its own seeded user and migration to be more HONEST; until then the choice is pinned here so + /// changing it is a decision rather than a diff. + /// + [Fact] + public void A_reclamation_is_attributed_to_the_system_background_actor() + { + SystemUsers.SeederId.ShouldBe(Guid.Parse("00000000-0000-0000-0000-000000000001")); + SystemUsers.SeederName.ShouldBe("System"); + } + [Fact] public void A_class_the_policy_does_not_register_has_no_rule_and_therefore_keeps() { diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs index 410e6158f..d5d46dd93 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs @@ -12,9 +12,10 @@ namespace CodeSpace.UnitTests.Workflows.Retention; /// these places, so the list IS the completeness argument, and a citer missing from it makes the cursor answer /// "unreferenced" about bytes something still reaches. /// -/// The drift detector below is the half that cannot be forgotten: it reads the EF model rather than this list, -/// so a future column that names an artifact object reds here even though nobody thought about retention while adding -/// it. +/// Two checks keep the list from being decoration. One reads the EF model rather than this list, so a future +/// column that names an artifact object reds here even though nobody thought about retention while adding it. The +/// other reads the cursor's own source, so an entry ADDED to the list without a probe beside it reds too — a pinned +/// list nothing cross-checks is a comment with a test around it. /// [Trait("Category", "Unit")] public sealed class LogStreamCitationSitesTests @@ -58,8 +59,53 @@ public void A_mapped_column_that_names_an_artifact_object_and_is_not_probed_fail } /// - /// A column names a CAS object when it ends in artifact_object_id. Three tables are excluded, each for a - /// stated reason rather than because it was inconvenient: + /// Every entry in the list has a probe beside it, checked against the cursor's own source: a site is only a site + /// if something asks it. The check is textual on purpose — the probes are LINQ over EF entities, so there is no + /// runtime handle to count, and the alternative (trusting the list) is what this test exists to refuse. + /// + [Fact] + public void Every_pinned_citation_site_is_actually_probed_by_the_cursor() + { + using var db = BuildContext(); + var source = File.ReadAllText(Path.Combine(ProductionSourceRoot(), "CodeSpace.Core", "Services", "Workflows", "Retention", "Cursors", $"{nameof(LogStreamRetentionCursor)}.cs")); + + var unprobed = LogStreamRetentionCursor.CitationSites + .Select(site => (site.Table, site.Column, Member: MemberOf(db, site.Table, site.Column))) + .Where(site => !source.Contains($".{site.Member}", StringComparison.Ordinal)) + .Select(site => $"{site.Table}.{site.Column} (no use of .{site.Member})") + .ToList(); + + unprobed.ShouldBeEmpty( + $"a site listed in {nameof(LogStreamRetentionCursor.CitationSites)} that nothing reads is a claim the cursor does not keep — " + + "add the probe, or take the entry out:\n " + string.Join("\n ", unprobed)); + } + + /// The CLR property a column is mapped from — the name the cursor's LINQ has to mention if it reads that column at all. + private static string MemberOf(CodeSpaceDbContext db, string table, string column) + { + var entity = db.Model.GetEntityTypes().FirstOrDefault(type => type.GetTableName() == table) + ?? throw new InvalidOperationException($"'{table}' is not mapped, so a citation site names a table this build does not have."); + + return (entity.GetProperties().FirstOrDefault(property => property.GetColumnName() == column) + ?? throw new InvalidOperationException($"'{table}.{column}' is not mapped, so a citation site names a column this build does not have.")).Name; + } + + private static string ProductionSourceRoot() + { + for (var directory = new DirectoryInfo(AppContext.BaseDirectory); directory is not null; directory = directory.Parent) + { + var candidate = Path.Combine(directory.FullName, "backend", "src"); + if (Directory.Exists(candidate)) return candidate; + } + + throw new DirectoryNotFoundException($"'backend/src' was not found above '{AppContext.BaseDirectory}', so the probes were never checked. Run the unit suite from the repository checkout."); + } + + /// + /// A column names a CAS object when it ends in artifact_object_id — which is a NAMING convention, not a + /// foreign key, so it catches a new column added in that shape and nothing else. A citer that named an object + /// under some other column name would pass this check; the pinned list above is what a reviewer reads for the + /// complete answer. Three tables are excluded, each for a stated reason rather than because it was inconvenient: /// /// artifact_object and artifact_location are the object's own identity and its placements, /// not a second holder of its bytes — the cursor reads both directly, and a purge is precisely what advances a diff --git a/frontend/src/api/agentRunLogsApi.test.ts b/frontend/src/api/agentRunLogsApi.test.ts index 1cfd838ee..c20caed8f 100644 --- a/frontend/src/api/agentRunLogsApi.test.ts +++ b/frontend/src/api/agentRunLogsApi.test.ts @@ -71,6 +71,18 @@ describe("Agent Run durable log API", () => { await expect(agentsApi.readRunLogRange("r", "missing", 0, 1)).resolves.toMatchObject({ availability: "Missing", isRetryable: false }); }); + // A reclaimed archive and a lost object both answer 410. If the union or the runtime Set omits Purged, the reader + // falls through to InvalidResponse and the operator is told the RESPONSE was malformed — strictly worse than the + // wrong-but-well-formed answer, and it hides a healthy deployment doing exactly what its retention policy says. + it("keeps the retention plane's Purged verdict distinct from a missing object", async () => { + vi.stubGlobal("fetch", vi.fn() + .mockResolvedValueOnce(json({ availability: "Purged", code: "log_bytes_purged", isRetryable: false, streamId: "s" }, 410)) + .mockResolvedValueOnce(json({ availability: "PhysicalObjectMissing", code: "artifact_missing", isRetryable: false, streamId: "s" }, 410))); + + await expect(agentsApi.readRunLogRange("r", "s", 0, 1)).resolves.toEqual({ availability: "Purged", code: "log_bytes_purged", isRetryable: false }); + await expect(agentsApi.readRunLogRange("r", "s", 0, 1)).resolves.toEqual({ availability: "PhysicalObjectMissing", code: "artifact_missing", isRetryable: false }); + }); + it("fails closed when a success response omits or contradicts its range contract", async () => { vi.stubGlobal("fetch", vi.fn(() => content(new Uint8Array([1, 2]), { "X-CodeSpace-Log-Offset": "0", diff --git a/frontend/src/api/agents.ts b/frontend/src/api/agents.ts index e72d35514..00f42da5b 100644 --- a/frontend/src/api/agents.ts +++ b/frontend/src/api/agents.ts @@ -246,6 +246,8 @@ export interface AgentRunLogStreamSummary { totalBytes: number; sha256: string | null; integrity?: AgentRunLogIntegrity | null; + /** When the retention plane reclaimed this stream's bytes. Set means the head row is a tombstone: the capture settled, the window elapsed, and `integrity` is deliberately absent because there is nothing left to verify. */ + purgedAt?: string | null; createdAt: string; lastModifiedAt: string; completedAt: string | null; @@ -257,7 +259,8 @@ export interface AgentRunLogPage { nextCursor: string | null; } -export type AgentRunLogReadAvailability = "InvalidRange" | "PhysicalObjectMissing" | "IntegrityFailure" | "BackendUnavailable" | "AccessDenied" | "ProviderTimeout" | "Unsupported"; +/** `Purged` is the retention plane's own answer: the capture settled and its bytes were later reclaimed on purpose. It is NOT `PhysicalObjectMissing`, which says an object that should still be there is gone. */ +export type AgentRunLogReadAvailability = "InvalidRange" | "PhysicalObjectMissing" | "IntegrityFailure" | "BackendUnavailable" | "AccessDenied" | "ProviderTimeout" | "Unsupported" | "Purged"; export interface AgentRunLogRangeAvailable { availability: "Available"; @@ -592,7 +595,7 @@ export const agentsApi = { fetchJson(`/api/agents/stats${since ? `?since=${encodeURIComponent(since)}` : ""}`), }; -const LOG_READ_AVAILABILITIES = new Set(["InvalidRange", "PhysicalObjectMissing", "IntegrityFailure", "BackendUnavailable", "AccessDenied", "ProviderTimeout", "Unsupported"]); +const LOG_READ_AVAILABILITIES = new Set(["InvalidRange", "PhysicalObjectMissing", "IntegrityFailure", "BackendUnavailable", "AccessDenied", "ProviderTimeout", "Unsupported", "Purged"]); const EVENT_DATA_READ_AVAILABILITIES = new Set(["NotReferenced", "InvalidRange", "MetadataMissing", "PhysicalObjectMissing", "IntegrityFailure", "BackendUnavailable", "AccessDenied"]); const AGENT_EVENT_KINDS = new Set(["Queued", "Started", "AssistantMessage", "Reasoning", "PlanUpdate", "ToolCall", "CommandExecuted", "FileChanged", "TestOutput", "ApprovalRequested", "ApprovalResolved", "Warning", "Error", "FinalSummary", "Completed"]); const LOG_STATUSES = new Set(["Open", "Completed", "Truncated", "Unavailable", "Corrupt", "CaptureFailed"]); diff --git a/frontend/src/components/workflows/AgentRunLogs.tsx b/frontend/src/components/workflows/AgentRunLogs.tsx index 5094c59f3..1b31ca9d4 100644 --- a/frontend/src/components/workflows/AgentRunLogs.tsx +++ b/frontend/src/components/workflows/AgentRunLogs.tsx @@ -323,6 +323,7 @@ function ReadProblem({ problem }: { problem: AgentRunLogRangeProblem }) { const title = (() => { switch (problem.availability) { case "Missing": case "PhysicalObjectMissing": return "Stored log bytes are missing"; + case "Purged": return "Log bytes were reclaimed by retention"; case "IntegrityFailure": return "Stored log bytes are corrupt"; case "BackendUnavailable": return "Storage backend unavailable"; case "AccessDenied": return "Storage access denied";