diff --git a/backend/src/CodeSpace.Api/Controllers/AgentsController.cs b/backend/src/CodeSpace.Api/Controllers/AgentsController.cs index c1850dfff..f443b118a 100644 --- a/backend/src/CodeSpace.Api/Controllers/AgentsController.cs +++ b/backend/src/CodeSpace.Api/Controllers/AgentsController.cs @@ -146,7 +146,9 @@ public async Task ReadRunLog([FromRoute] Guid agentRunId, [FromRo AgentRunLogReadAvailability.InvalidRange => StatusCodes.Status400BadRequest, AgentRunLogReadAvailability.AccessDenied => StatusCodes.Status424FailedDependency, AgentRunLogReadAvailability.BackendUnavailable or AgentRunLogReadAvailability.ProviderTimeout => StatusCodes.Status503ServiceUnavailable, - AgentRunLogReadAvailability.PhysicalObjectMissing or AgentRunLogReadAvailability.IntegrityFailure => StatusCodes.Status410Gone, + // 410 for all three: the bytes are not coming back. The availability code beside it is what distinguishes a + // policy reclamation from a loss, which the status alone cannot say. + AgentRunLogReadAvailability.PhysicalObjectMissing or AgentRunLogReadAvailability.IntegrityFailure or AgentRunLogReadAvailability.Purged => StatusCodes.Status410Gone, _ => StatusCodes.Status422UnprocessableEntity, }; diff --git a/backend/src/CodeSpace.Core/Handlers/CommandHandlers/Workflows/ReapExpiredDurableRecordsCommandHandler.cs b/backend/src/CodeSpace.Core/Handlers/CommandHandlers/Workflows/ReapExpiredDurableRecordsCommandHandler.cs new file mode 100644 index 000000000..5132c4c7b --- /dev/null +++ b/backend/src/CodeSpace.Core/Handlers/CommandHandlers/Workflows/ReapExpiredDurableRecordsCommandHandler.cs @@ -0,0 +1,20 @@ +using CodeSpace.Core.Services.Workflows.Retention; +using CodeSpace.Messages.Commands.Workflows; +using MediatR; + +namespace CodeSpace.Core.Handlers.CommandHandlers.Workflows; + +/// Rule 16 — thin handler. The bounded sweep lives in . +public sealed class ReapExpiredDurableRecordsCommandHandler : IRequestHandler +{ + private readonly IDurableRetentionReaper _reaper; + + public ReapExpiredDurableRecordsCommandHandler(IDurableRetentionReaper reaper) { _reaper = reaper; } + + public async Task Handle(ReapExpiredDurableRecordsCommand request, CancellationToken cancellationToken) + { + var summary = await _reaper.SweepAsync(cancellationToken).ConfigureAwait(false); + + return new ReapExpiredDurableRecordsResponse { Summary = summary }; + } +} diff --git a/backend/src/CodeSpace.Core/Handlers/QueryHandlers/Agents/AgentRunLogQueryHandlers.cs b/backend/src/CodeSpace.Core/Handlers/QueryHandlers/Agents/AgentRunLogQueryHandlers.cs index 6be8ef105..e4ca9bd26 100644 --- a/backend/src/CodeSpace.Core/Handlers/QueryHandlers/Agents/AgentRunLogQueryHandlers.cs +++ b/backend/src/CodeSpace.Core/Handlers/QueryHandlers/Agents/AgentRunLogQueryHandlers.cs @@ -208,6 +208,7 @@ public static AgentRunLogRangeRead Available(AgentRunLogRangeResult.Available va { AgentRunLogProblemCode.InvalidRequest => AgentRunLogReadAvailability.InvalidRange, AgentRunLogProblemCode.Missing or AgentRunLogProblemCode.ArtifactMissing => AgentRunLogReadAvailability.PhysicalObjectMissing, + AgentRunLogProblemCode.Purged => AgentRunLogReadAvailability.Purged, AgentRunLogProblemCode.ArtifactCorrupt => AgentRunLogReadAvailability.IntegrityFailure, AgentRunLogProblemCode.AccessDenied => AgentRunLogReadAvailability.AccessDenied, AgentRunLogProblemCode.ProviderTimeout => AgentRunLogReadAvailability.ProviderTimeout, diff --git a/backend/src/CodeSpace.Core/Jobs/RecurringJobs/DurableRetentionReaperRecurringJob.cs b/backend/src/CodeSpace.Core/Jobs/RecurringJobs/DurableRetentionReaperRecurringJob.cs new file mode 100644 index 000000000..5ba0173ed --- /dev/null +++ b/backend/src/CodeSpace.Core/Jobs/RecurringJobs/DurableRetentionReaperRecurringJob.cs @@ -0,0 +1,25 @@ +using CodeSpace.Messages.Commands.Workflows; +using MediatR; + +namespace CodeSpace.Core.Jobs.RecurringJobs; + +/// +/// Hourly, dispatches to reclaim durable records nothing cites any +/// more. Thin Mediator dispatcher (Rule 14) — the work lives in +/// . +/// +/// Hourly is ample and deliberately unhurried: the shortest rule keeps a record for days before it is even a +/// candidate and then quarantines it for another day, so the cadence changes nothing about WHAT is reclaimed — only +/// how promptly. Offset to :45 so it does not pile onto the :15 artifact reaper or the :30 spool reaper. +/// +public sealed class DurableRetentionReaperRecurringJob : IRecurringJob +{ + private readonly IMediator _mediator; + + public DurableRetentionReaperRecurringJob(IMediator mediator) { _mediator = mediator; } + + public string JobId => nameof(DurableRetentionReaperRecurringJob); + public string CronExpression => "45 * * * *"; // quarter to every hour + + public async Task Execute() => await _mediator.Send(new ReapExpiredDurableRecordsCommand()).ConfigureAwait(false); +} diff --git a/backend/src/CodeSpace.Core/Persistence/Db/CodeSpaceDbContext.cs b/backend/src/CodeSpace.Core/Persistence/Db/CodeSpaceDbContext.cs index bc64aed63..bae1ef5c0 100644 --- a/backend/src/CodeSpace.Core/Persistence/Db/CodeSpaceDbContext.cs +++ b/backend/src/CodeSpace.Core/Persistence/Db/CodeSpaceDbContext.cs @@ -117,6 +117,7 @@ public CodeSpaceDbContext(DbContextOptions options, ICurrent public DbSet PairedQualificationCellAdmission => Set(); public DbSet PairedQualificationCellCheckpoint => Set(); public DbSet PairedQualificationResult => Set(); + public DbSet PairedQualificationResultPin => Set(); public DbSet Lesson => Set(); public DbSet SupervisorDecisionRecord => Set(); diff --git a/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0235_durable_retention_pins.sql b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0235_durable_retention_pins.sql new file mode 100644 index 000000000..66fcdf7f4 --- /dev/null +++ b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0235_durable_retention_pins.sql @@ -0,0 +1,114 @@ +-- 0235_durable_retention_pins.sql +-- +-- Two things nothing in the schema could say before: which durable records a sealed qualification result still cites, +-- and when a log stream's bytes became reclaimable. +-- +-- WHY A PIN TABLE. A qualification result is an immutable statement ABOUT a set of agent runs — their logs, their +-- cleanup receipts, their offloaded event payloads. None of that is expressed as a foreign key anywhere: the result row +-- carries digests and a JSON outcome, and the observations it was computed from name their run in benchmark_result. +-- So "is this log stream still cited by a sealed result" was a question no reaper could ask, and a reaper that cannot +-- ask it must never delete. The pin is the answer, written in the SAME transaction as the seal: if the seal commits, +-- every record it cites is pinned, and if it does not, no pin exists either. +-- +-- WHY TWO TARGET COLUMNS. ArtifactReferenceOracle enumerates every column in the schema whose name ends in +-- `artifact_id` and whose target is workflow_artifact — that naming rule IS its completeness argument. An artifact pin +-- must therefore live in a column called pinned_artifact_id and nothing else may, or the oracle would read a log +-- stream id as an artifact reference. Every other pinned kind uses pinned_id. The CHECK ties the two together so the +-- pair can never disagree with the kind. +-- +-- The uniqueness the pin needs is over (result, kind, target), which spans both nullable columns, so it is an +-- expression index rather than the primary key; the surrogate id keeps the row addressable like every other entity. +-- +-- WHY retain_until AND purged_at. A log stream's bytes live in routed CAS objects that nothing ever reclaimed. The +-- reaper's two waits need somewhere durable to stand: retain_until is the instant the bytes become reclaimable — the +-- reaper writes it on the FIRST sweep that finds the stream terminal, past its rule and cited by nobody, and collects +-- only on a later sweep once it has passed. purged_at is the tombstone. The stream head row deliberately SURVIVES its +-- bytes: a reader that finds no row cannot tell "purged by policy" from "lost", and the whole point of a retention +-- plane is that reclamation is legible. Both are metadata-only nullable ADD COLUMNs (no rewrite, no index), and NULL +-- is the correct reading for every row an older binary writes or reads. +-- +-- The rules that keep these two columns honest are in the guard function, not in a CHECK: see +-- 0236_agent_run_log_stream_retention.sql. A CHECK here would have to be validated against the whole table. +-- +-- Rollback: DROP TABLE paired_qualification_result_pin; ALTER TABLE agent_run_log_stream DROP COLUMN retain_until, +-- DROP COLUMN purged_at. Nothing reads either from an older binary. + +CREATE TABLE paired_qualification_result_pin ( + id uuid PRIMARY KEY, + result_id uuid NOT NULL REFERENCES paired_qualification_result(observation_group_id) ON DELETE CASCADE, + kind varchar(32) NOT NULL, + pinned_id uuid NULL, + pinned_artifact_id uuid NULL, + pinned_at timestamptz NOT NULL, + CONSTRAINT ck_paired_qualification_result_pin_kind + CHECK (kind IN ('LogStream', 'Artifact', 'CleanupReceipt', 'AgentRun')), + CONSTRAINT ck_paired_qualification_result_pin_target + CHECK (num_nonnulls(pinned_id, pinned_artifact_id) = 1 AND (kind = 'Artifact') = (pinned_artifact_id IS NOT NULL)) +); + +CREATE UNIQUE INDEX ux_paired_qualification_result_pin + ON paired_qualification_result_pin (result_id, kind, COALESCE(pinned_id, pinned_artifact_id)); + +-- One index per probed column, partial so each is only the rows that column addresses: every consumer asks +-- "EXISTS (SELECT 1 ... WHERE = $1)", and `= $1` implies NOT NULL, so the partial index serves it. +CREATE INDEX ix_paired_qualification_result_pin_target + ON paired_qualification_result_pin (pinned_id) WHERE pinned_id IS NOT NULL; +CREATE INDEX ix_paired_qualification_result_pin_artifact + ON paired_qualification_result_pin (pinned_artifact_id) WHERE pinned_artifact_id IS NOT NULL; + +COMMENT ON TABLE paired_qualification_result_pin IS + 'Which durable records a sealed paired-qualification result cites. Written in the seal''s own transaction; read by every retention cursor as the "still referenced" proof.'; + +-- Backfill, and it is not optional. A pin table that only fills going forward would make every result sealed before +-- this migration read as citing NOTHING, and the first reaper sweep would then find its evidence collectable. The four +-- statements below are the same four closures PairedQualificationResultStore writes for a new seal, in SQL: the runs a +-- result's observations name, and then that run's log streams, cleanup receipts and offloaded event payloads. They and +-- the writer are pinned against each other by an integration test. + +INSERT INTO paired_qualification_result_pin (id, result_id, kind, pinned_id, pinned_artifact_id, pinned_at) +SELECT gen_random_uuid(), cited.result_id, 'AgentRun', cited.agent_run_id, NULL, now() +FROM ( + SELECT DISTINCT result.observation_group_id AS result_id, observation.agent_run_id + FROM paired_qualification_result result + JOIN benchmark_result observation ON observation.observation_group_id = result.observation_group_id + WHERE observation.agent_run_id IS NOT NULL +) AS cited +ON CONFLICT DO NOTHING; + +INSERT INTO paired_qualification_result_pin (id, result_id, kind, pinned_id, pinned_artifact_id, pinned_at) +SELECT gen_random_uuid(), cited.result_id, 'LogStream', cited.stream_id, NULL, now() +FROM ( + SELECT DISTINCT pin.result_id, stream.id AS stream_id + FROM paired_qualification_result_pin pin + JOIN agent_run_log_stream stream ON stream.agent_run_id = pin.pinned_id + WHERE pin.kind = 'AgentRun' +) AS cited +ON CONFLICT DO NOTHING; + +INSERT INTO paired_qualification_result_pin (id, result_id, kind, pinned_id, pinned_artifact_id, pinned_at) +SELECT gen_random_uuid(), cited.result_id, 'CleanupReceipt', cited.receipt_id, NULL, now() +FROM ( + SELECT DISTINCT pin.result_id, receipt.id AS receipt_id + FROM paired_qualification_result_pin pin + JOIN agent_run_cleanup_receipt receipt ON receipt.agent_run_id = pin.pinned_id + WHERE pin.kind = 'AgentRun' +) AS cited +ON CONFLICT DO NOTHING; + +INSERT INTO paired_qualification_result_pin (id, result_id, kind, pinned_id, pinned_artifact_id, pinned_at) +SELECT gen_random_uuid(), cited.result_id, 'Artifact', NULL, cited.data_artifact_id, now() +FROM ( + SELECT DISTINCT pin.result_id, event.data_artifact_id + FROM paired_qualification_result_pin pin + JOIN agent_run_event event ON event.agent_run_id = pin.pinned_id + WHERE pin.kind = 'AgentRun' AND event.data_artifact_id IS NOT NULL +) AS cited +ON CONFLICT DO NOTHING; + +ALTER TABLE agent_run_log_stream ADD COLUMN retain_until timestamptz NULL; +ALTER TABLE agent_run_log_stream ADD COLUMN purged_at timestamptz NULL; + +COMMENT ON COLUMN agent_run_log_stream.retain_until IS + 'The earliest instant this stream''s bytes may be reclaimed, written on the first sweep that found it collectable. NULL means no sweep has ever proposed it.'; +COMMENT ON COLUMN agent_run_log_stream.purged_at IS + 'When this stream''s segment bytes were reclaimed. The head row survives as the tombstone so a reader reads "purged", never a missing stream.'; diff --git a/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql new file mode 100644 index 000000000..e66a796f7 --- /dev/null +++ b/backend/src/CodeSpace.Core/Persistence/DbUpFiles/0236_agent_run_log_stream_retention.sql @@ -0,0 +1,305 @@ +-- 0236_agent_run_log_stream_retention.sql +-- +-- A terminal log stream could not be told that its bytes are gone. +-- +-- agent_run_log_stream_guard() admits five update shapes (a capture claim, a source finalization, a remote-stall +-- statement, an exact one-segment head advance, a terminal transition) and refuses everything else outright: +-- +-- IF OLD.state <> 'Open' THEN RAISE EXCEPTION 'agent_run_log_stream terminal state is immutable (id=%).' ... +-- +-- That is right for every statement a capturing worker makes, and it is kept for all of them. But it also means the +-- retention columns 0235 added can never be written: a stream is a retention candidate only once it is terminal, and +-- from that moment the row admits no update at all. DELETE is refused too, and deliberately so — the head row is the +-- tombstone that lets a reader say "purged" instead of finding nothing. +-- +-- This migration adds ONE admissible shape, the sixth, and makes it as narrow as the row allows: a RETENTION +-- STATEMENT may move retain_until and purged_at on a terminal stream and NOTHING else. It cannot change the state, the +-- claim, the byte head, the source offsets, the finalization receipt, the digests or the error — so a purge can never +-- be mistaken for a capture verdict, and the durable prefix's metadata stays exactly as readable as it was. A +-- statement that touches anything else still reads the refusal it always did, word for word, because the arm +-- reproduces it rather than replacing it. Three further rules ride with the arm, each of which exists because its +-- absence is silent data loss: +-- +-- * A stream that is still Open is never a retention candidate. Its bytes belong to a live capture session, so the +-- columns are refused there rather than left to fall through an arm that does not enumerate them. +-- * A purge requires a retain_until that was recorded by an EARLIER statement (OLD, not NEW). The reaper's two +-- waits are an age floor and then a quarantine; without this rule one sweep could both propose and execute a +-- collection in a single statement, and the second wait would never have existed at all. +-- * A purge is final. Clearing purged_at would claim bytes are back that no one restored. +-- +-- The arm matches on the SHAPE of the statement rather than on "one of the two columns changed value", because a +-- sweep that looked at a stream and kept it must be able to record that it looked — advancing the revision and the +-- modification time while both columns keep the values they had. Without that write the claim query would hand the +-- same unreclaimable rows back on every tick, for ever. +-- +-- The INSERT arm is tightened for the same reason it rejects a pre-set manifest receipt: a new stream that arrives +-- already carrying a retention verdict is not a new stream. +-- +-- Supersedes 0232_agent_run_log_stream_owner_loss.sql, which is reproduced verbatim apart from the arm above and the +-- INSERT clause. The function has been redefined at 0129, 0132, 0133, 0201, 0230 and 0232 — one number, one +-- definition, every time (MigrationDiscoveryTests.No_two_migrations_at_one_number_redefine_the_same_function), because +-- DbUp is FILENAME-keyed: two files at one number both run and the later NAME silently wins. +-- +-- No DDL, no new column, no new trigger. +-- +-- Rollback: re-run 0232_agent_run_log_stream_owner_loss.sql's CREATE OR REPLACE FUNCTION body. Any stream already +-- carrying purged_at keeps it; nothing can write another one. + +CREATE OR REPLACE FUNCTION agent_run_log_stream_guard() RETURNS trigger AS $$ +DECLARE + appended agent_run_log_segment%ROWTYPE; + current_fence BIGINT; + is_claim BOOLEAN := FALSE; + is_source_finalize BOOLEAN := FALSE; + is_remote_stall BOOLEAN := FALSE; + is_owner_loss BOOLEAN := FALSE; +BEGIN + IF TG_OP = 'DELETE' THEN + RAISE EXCEPTION 'agent_run_log_stream is durable capture state — DELETE rejected (id=%).', OLD.id; + END IF; + + IF TG_OP = 'INSERT' THEN + IF NEW.manifest_digest IS NOT NULL THEN + RAISE EXCEPTION 'A new log stream cannot carry a manifest receipt.'; + END IF; + SELECT fence_epoch INTO current_fence FROM agent_run + WHERE team_id = NEW.team_id AND id = NEW.agent_run_id + FOR SHARE; + IF NOT FOUND OR current_fence <= 0 OR NEW.worker_fence_epoch IS DISTINCT FROM current_fence + OR NEW.capture_session_id IS NULL OR NEW.schema_version <> 3 THEN + RAISE EXCEPTION 'agent_run_log_stream requires the current positive AgentRun fence and a capture session (run_id=%, attempted_fence=%).', NEW.agent_run_id, NEW.worker_fence_epoch; + END IF; + IF NEW.state <> 'Open' OR NEW.revision <> 1 OR NEW.segment_count <> 0 OR NEW.total_bytes <> 0 + OR NEW.source_offset_bytes <> 0 OR NEW.capture_source_base_offset_bytes <> 0 + OR NEW.capture_finalized_at IS NOT NULL OR NEW.next_segment_ordinal <> 1 OR NEW.next_offset_bytes <> 0 + OR NEW.completed_at IS NOT NULL OR NEW.error_code IS NOT NULL OR NEW.error_message IS NOT NULL + OR NEW.content_digest IS NOT NULL OR NEW.content_digest_algorithm IS NOT NULL + OR NEW.remote_stall_since IS NOT NULL OR NEW.remote_stall_code IS NOT NULL + OR NEW.retain_until IS NOT NULL OR NEW.purged_at IS NOT NULL + OR NEW.last_modified_at < NEW.created_at THEN + RAISE EXCEPTION 'agent_run_log_stream must start as an empty Open revision-one head (id=%).', NEW.id; + END IF; + RETURN NEW; + END IF; + + -- The SIXTH admissible update shape, and the reason this function is redefined: a RETENTION STATEMENT. It is + -- matched first and answers EVERY update to a terminal row, because the arms below are written for statements a + -- capturing worker makes and each one refuses a terminal row outright. The refusal they used to reach is + -- reproduced here word for word, so a caller trying to revive or rewrite a settled stream still reads exactly the + -- message it always did. + -- + -- The shape, not a value change, is what admits it: a retention statement may touch retain_until and purged_at, + -- must advance the revision and the modification time, and may touch NOTHING else. Matching on the shape rather + -- than on "one of the two columns differs" is deliberate — a sweep that looked at a stream and KEPT it has to be + -- able to record that it looked, and that statement changes neither column's value. + IF OLD.state <> 'Open' THEN + IF NEW.id IS DISTINCT FROM OLD.id OR NEW.team_id IS DISTINCT FROM OLD.team_id + OR NEW.agent_run_id IS DISTINCT FROM OLD.agent_run_id OR NEW.state IS DISTINCT FROM OLD.state + OR NEW.stream_kind IS DISTINCT FROM OLD.stream_kind OR NEW.content_type IS DISTINCT FROM OLD.content_type + OR NEW.content_encoding IS DISTINCT FROM OLD.content_encoding OR NEW.capture_source IS DISTINCT FROM OLD.capture_source + OR NEW.schema_version IS DISTINCT FROM OLD.schema_version OR NEW.retention IS DISTINCT FROM OLD.retention + OR NEW.expires_at IS DISTINCT FROM OLD.expires_at OR NEW.created_at IS DISTINCT FROM OLD.created_at + OR NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id + OR NEW.segment_count IS DISTINCT FROM OLD.segment_count OR NEW.total_bytes IS DISTINCT FROM OLD.total_bytes + OR NEW.source_offset_bytes IS DISTINCT FROM OLD.source_offset_bytes + OR NEW.capture_source_base_offset_bytes IS DISTINCT FROM OLD.capture_source_base_offset_bytes + OR NEW.next_segment_ordinal IS DISTINCT FROM OLD.next_segment_ordinal + OR NEW.next_offset_bytes IS DISTINCT FROM OLD.next_offset_bytes + OR NEW.capture_finalized_at IS DISTINCT FROM OLD.capture_finalized_at + OR NEW.completed_at IS DISTINCT FROM OLD.completed_at + OR NEW.error_code IS DISTINCT FROM OLD.error_code OR NEW.error_message IS DISTINCT FROM OLD.error_message + OR NEW.content_digest_algorithm IS DISTINCT FROM OLD.content_digest_algorithm + OR NEW.content_digest IS DISTINCT FROM OLD.content_digest + OR NEW.manifest_digest IS DISTINCT FROM OLD.manifest_digest + OR NEW.remote_stall_since IS DISTINCT FROM OLD.remote_stall_since + OR NEW.remote_stall_code IS DISTINCT FROM OLD.remote_stall_code THEN + RAISE EXCEPTION 'agent_run_log_stream terminal state is immutable (id=%, state=%).', OLD.id, OLD.state; + END IF; + IF OLD.purged_at IS NOT NULL THEN + RAISE EXCEPTION 'agent_run_log_stream purge is final; a purged stream admits no further retention statement (id=%).', OLD.id; + END IF; + -- OLD, not NEW: the quarantine must have been recorded by an EARLIER statement, or one sweep could both + -- propose a collection and carry it out, and the second of the two waits would never have existed. + IF NEW.purged_at IS NOT NULL AND OLD.retain_until IS NULL THEN + RAISE EXCEPTION 'agent_run_log_stream cannot be purged without the retain_until it was quarantined under (id=%).', OLD.id; + END IF; + IF NEW.revision <> OLD.revision + 1 OR NEW.last_modified_at < OLD.last_modified_at THEN + RAISE EXCEPTION 'agent_run_log_stream revision/time must advance monotonically (id=%, old_revision=%, new_revision=%).', OLD.id, OLD.revision, NEW.revision; + END IF; + RETURN NEW; + END IF; + + IF NEW.retain_until IS DISTINCT FROM OLD.retain_until OR NEW.purged_at IS DISTINCT FROM OLD.purged_at THEN + RAISE EXCEPTION 'agent_run_log_stream retention statement rejected on a live stream (id=%); its bytes belong to an open capture session.', OLD.id; + END IF; + + IF NEW.manifest_digest IS DISTINCT FROM OLD.manifest_digest + AND NOT (OLD.state = 'Open' AND NEW.state = 'Completed' AND NEW.schema_version = 3) THEN + RAISE EXCEPTION 'A log manifest receipt may only be set by v3 completion.'; + END IF; + IF NEW.schema_version = 3 AND NEW.state = 'Completed' THEN + IF NEW.content_digest IS NOT NULL OR NEW.content_digest_algorithm IS NOT NULL OR NEW.manifest_digest IS NULL + OR NOT EXISTS ( + SELECT 1 FROM agent_run_log_verification verification + WHERE verification.team_id = NEW.team_id AND verification.agent_run_id = NEW.agent_run_id + AND verification.stream_id = NEW.id AND verification.stream_revision = OLD.revision + AND verification.worker_fence_epoch = NEW.worker_fence_epoch AND verification.capture_session_id = NEW.capture_session_id + AND verification.segment_count = NEW.segment_count AND verification.total_bytes = NEW.total_bytes + AND verification.source_offset_bytes = NEW.source_offset_bytes AND verification.sealed_at IS NOT NULL + AND verification.next_segment_ordinal = NEW.segment_count + 1 AND verification.verified_bytes = NEW.total_bytes + AND verification.manifest_digest = NEW.manifest_digest) THEN + RAISE EXCEPTION 'v3 completion requires its exact sealed segment manifest; a whole SHA is not a manifest receipt.'; + END IF; + END IF; + + IF NEW.id IS DISTINCT FROM OLD.id OR NEW.team_id IS DISTINCT FROM OLD.team_id + OR NEW.agent_run_id IS DISTINCT FROM OLD.agent_run_id OR NEW.stream_kind IS DISTINCT FROM OLD.stream_kind + OR NEW.content_type IS DISTINCT FROM OLD.content_type OR NEW.content_encoding IS DISTINCT FROM OLD.content_encoding + OR NEW.capture_source IS DISTINCT FROM OLD.capture_source OR NEW.schema_version IS DISTINCT FROM OLD.schema_version + OR NEW.retention IS DISTINCT FROM OLD.retention OR NEW.expires_at IS DISTINCT FROM OLD.expires_at + OR NEW.created_at IS DISTINCT FROM OLD.created_at THEN + RAISE EXCEPTION 'agent_run_log_stream stable identity is immutable (id=%).', OLD.id; + END IF; + IF NEW.revision <> OLD.revision + 1 OR NEW.last_modified_at < OLD.last_modified_at THEN + RAISE EXCEPTION 'agent_run_log_stream revision/time must advance monotonically (id=%, old_revision=%, new_revision=%).', OLD.id, OLD.revision, NEW.revision; + END IF; + + IF NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id THEN + SELECT fence_epoch INTO current_fence FROM agent_run + WHERE team_id = NEW.team_id AND id = NEW.agent_run_id + FOR SHARE; + IF NOT FOUND OR current_fence <= 0 OR NEW.worker_fence_epoch IS DISTINCT FROM current_fence + OR NEW.worker_fence_epoch < COALESCE(OLD.worker_fence_epoch, 0) OR NEW.capture_session_id IS NULL THEN + RAISE EXCEPTION 'agent_run_log_stream stale or malformed capture claim rejected (run_id=%, current=%, attempted=%).', NEW.agent_run_id, current_fence, NEW.worker_fence_epoch; + END IF; + IF NEW.capture_session_id IS NOT DISTINCT FROM OLD.capture_session_id THEN + IF NEW.worker_fence_epoch <= COALESCE(OLD.worker_fence_epoch, 0) + OR NEW.capture_source_base_offset_bytes IS DISTINCT FROM OLD.capture_source_base_offset_bytes + OR NEW.capture_finalized_at IS DISTINCT FROM OLD.capture_finalized_at THEN + RAISE EXCEPTION 'agent_run_log_stream same-session reclaim requires a strictly newer fence and preserves source state (id=%).', OLD.id; + END IF; + ELSIF OLD.capture_finalized_at IS NULL OR NEW.capture_source_base_offset_bytes <> OLD.source_offset_bytes + OR NEW.capture_finalized_at IS NOT NULL THEN + RAISE EXCEPTION 'agent_run_log_stream stale or malformed capture claim rejected: next spool requires a finalized prior source and starts at its source head (id=%).', OLD.id; + END IF; + IF NEW.state IS DISTINCT FROM OLD.state OR NEW.segment_count IS DISTINCT FROM OLD.segment_count + OR NEW.total_bytes IS DISTINCT FROM OLD.total_bytes OR NEW.source_offset_bytes IS DISTINCT FROM OLD.source_offset_bytes + OR NEW.next_segment_ordinal IS DISTINCT FROM OLD.next_segment_ordinal + OR NEW.next_offset_bytes IS DISTINCT FROM OLD.next_offset_bytes OR NEW.completed_at IS DISTINCT FROM OLD.completed_at + OR NEW.error_code IS DISTINCT FROM OLD.error_code OR NEW.error_message IS DISTINCT FROM OLD.error_message + OR NEW.content_digest_algorithm IS DISTINCT FROM OLD.content_digest_algorithm + OR NEW.content_digest IS DISTINCT FROM OLD.content_digest THEN + RAISE EXCEPTION 'agent_run_log_stream capture claim cannot mutate byte or terminal state (id=%).', OLD.id; + END IF; + -- The two stall columns are deliberately ABSENT from that list: a claim is the one statement that may clear + -- them. A marker is one producer's statement about a segment it holds in memory, and a claim supersedes that + -- producer -- so a marker the claim inherited would have no one left to clear it and every reader would call + -- a healthily-capturing stream stalled forever. Do not add them here. + is_claim := TRUE; + END IF; + + IF NOT is_claim AND NEW.state = 'Open' AND OLD.capture_finalized_at IS NULL + AND NEW.capture_finalized_at IS NOT NULL THEN + IF NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id + OR NEW.segment_count IS DISTINCT FROM OLD.segment_count OR NEW.total_bytes IS DISTINCT FROM OLD.total_bytes + OR NEW.source_offset_bytes IS DISTINCT FROM OLD.source_offset_bytes + OR NEW.capture_source_base_offset_bytes IS DISTINCT FROM OLD.capture_source_base_offset_bytes + OR NEW.next_segment_ordinal IS DISTINCT FROM OLD.next_segment_ordinal + OR NEW.next_offset_bytes IS DISTINCT FROM OLD.next_offset_bytes OR NEW.completed_at IS NOT NULL + OR NEW.error_code IS NOT NULL OR NEW.error_message IS NOT NULL + OR NEW.content_digest_algorithm IS NOT NULL OR NEW.content_digest IS NOT NULL THEN + RAISE EXCEPTION 'agent_run_log_stream source finalization cannot rewrite its claim or byte head (id=%).', OLD.id; + END IF; + is_source_finalize := TRUE; + END IF; + + -- The FIFTH admissible update shape, and the reason this function is redefined: a REMOTE-STALL statement. The + -- four shapes above are a claim, a source finalization, an exact one-segment head advance, and a terminal + -- transition -- so a producer holding a segment behind a transient outage had no legal way to SAY so, and its + -- statement was refused as an append it never made. This arm admits exactly that statement and nothing more: the + -- two stall columns move, the revision advances as every update here must, and every claim, byte-head and + -- terminal column is required to be untouched. That is what keeps "my bytes are queued" from ever being + -- mistakable for "my bytes are stored". + IF NOT is_claim AND NOT is_source_finalize AND NEW.state = 'Open' + AND (NEW.remote_stall_since IS DISTINCT FROM OLD.remote_stall_since + OR NEW.remote_stall_code IS DISTINCT FROM OLD.remote_stall_code) THEN + IF NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id + OR NEW.segment_count <> OLD.segment_count OR NEW.total_bytes <> OLD.total_bytes + OR NEW.source_offset_bytes <> OLD.source_offset_bytes + OR NEW.capture_source_base_offset_bytes <> OLD.capture_source_base_offset_bytes + OR NEW.capture_finalized_at IS DISTINCT FROM OLD.capture_finalized_at + OR NEW.next_segment_ordinal <> OLD.next_segment_ordinal OR NEW.next_offset_bytes <> OLD.next_offset_bytes + OR NEW.completed_at IS NOT NULL OR NEW.error_code IS NOT NULL OR NEW.error_message IS NOT NULL + OR NEW.content_digest_algorithm IS NOT NULL OR NEW.content_digest IS NOT NULL + OR NEW.manifest_digest IS DISTINCT FROM OLD.manifest_digest THEN + RAISE EXCEPTION 'agent_run_log_stream remote-stall statement cannot rewrite its claim, byte head or terminal state (id=%).', OLD.id; + END IF; + is_remote_stall := TRUE; + END IF; + + IF NOT is_claim AND NOT is_source_finalize AND NOT is_remote_stall AND NEW.state = 'Open' THEN + IF NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id + OR OLD.capture_finalized_at IS NOT NULL OR NEW.capture_finalized_at IS NOT NULL + OR NEW.capture_source_base_offset_bytes IS DISTINCT FROM OLD.capture_source_base_offset_bytes + OR NEW.segment_count <> OLD.segment_count + 1 OR NEW.next_segment_ordinal <> OLD.next_segment_ordinal + 1 + OR NEW.total_bytes <= OLD.total_bytes OR NEW.next_offset_bytes <> NEW.total_bytes + OR NEW.source_offset_bytes <= OLD.source_offset_bytes + OR NEW.completed_at IS NOT NULL OR NEW.error_code IS NOT NULL OR NEW.error_message IS NOT NULL + OR NEW.content_digest_algorithm IS NOT NULL OR NEW.content_digest IS NOT NULL THEN + RAISE EXCEPTION 'agent_run_log_stream Open updates are exact one-segment head advances (id=%).', OLD.id; + END IF; + + SELECT * INTO appended FROM agent_run_log_segment + WHERE team_id = NEW.team_id AND stream_id = NEW.id AND agent_run_id = NEW.agent_run_id + AND segment_ordinal = OLD.next_segment_ordinal; + IF NOT FOUND OR appended.start_offset_bytes <> OLD.next_offset_bytes + OR appended.length_bytes <> NEW.total_bytes - OLD.total_bytes + OR appended.source_start_offset_bytes <> OLD.source_offset_bytes + OR appended.source_length_bytes <> NEW.source_offset_bytes - OLD.source_offset_bytes + OR appended.worker_fence_epoch IS DISTINCT FROM NEW.worker_fence_epoch + OR appended.capture_session_id IS DISTINCT FROM NEW.capture_session_id + OR appended.created_at > NEW.last_modified_at THEN + RAISE EXCEPTION 'agent_run_log_stream head advance requires its exact claimed append-only segment (id=%, ordinal=%).', OLD.id, OLD.next_segment_ordinal; + END IF; + ELSIF NOT is_claim AND NOT is_source_finalize AND NOT is_remote_stall THEN + SELECT fence_epoch INTO current_fence FROM agent_run + WHERE team_id = NEW.team_id AND id = NEW.agent_run_id + FOR SHARE; + IF NOT FOUND OR current_fence <= 0 THEN + RAISE EXCEPTION 'agent_run_log_stream terminal transition requires a live run fence (run_id=%, current=%).', NEW.agent_run_id, current_fence; + END IF; + -- The owner-loss exception, and the whole of it. A stream whose fence is STRICTLY behind the run's belongs to + -- a generation that is provably over, so no live capture session can still be appending to it and the only + -- thing left to say about it is that its owner is gone. CaptureFailed is the one state that says that; every + -- other terminal state is a claim about the bytes, which a generation that never read them cannot make. + is_owner_loss := OLD.worker_fence_epoch < current_fence AND NEW.state = 'CaptureFailed'; + IF NOT is_owner_loss AND OLD.worker_fence_epoch IS DISTINCT FROM current_fence THEN + RAISE EXCEPTION 'agent_run_log_stream terminal transition requires its current worker fence (run_id=%, current=%, claimed=%).', NEW.agent_run_id, current_fence, OLD.worker_fence_epoch; + END IF; + IF NEW.worker_fence_epoch IS DISTINCT FROM OLD.worker_fence_epoch + OR NEW.capture_session_id IS DISTINCT FROM OLD.capture_session_id + OR NEW.segment_count <> OLD.segment_count OR NEW.total_bytes <> OLD.total_bytes + OR NEW.source_offset_bytes <> OLD.source_offset_bytes + OR NEW.capture_source_base_offset_bytes <> OLD.capture_source_base_offset_bytes + OR NEW.capture_finalized_at IS DISTINCT FROM OLD.capture_finalized_at + OR NEW.next_segment_ordinal <> OLD.next_segment_ordinal OR NEW.next_offset_bytes <> OLD.next_offset_bytes + OR NEW.completed_at IS NULL OR NEW.completed_at < OLD.created_at THEN + RAISE EXCEPTION 'agent_run_log_stream terminal transition cannot rewrite its claim or byte head (id=%).', OLD.id; + END IF; + IF NEW.state = 'Completed' AND NEW.schema_version <> 3 AND (NEW.content_digest_algorithm IS DISTINCT FROM 'Sha256' + OR NEW.content_digest IS NULL OR octet_length(NEW.content_digest) IS DISTINCT FROM 32) THEN + RAISE EXCEPTION 'agent_run_log_stream Completed requires its verified SHA-256 content digest (id=%).', OLD.id; + END IF; + IF NEW.state = 'Completed' AND OLD.capture_finalized_at IS NULL THEN + RAISE EXCEPTION 'agent_run_log_stream Completed requires a durable final-drain receipt (id=%).', OLD.id; + END IF; + END IF; + + RETURN NEW; +END; +$$ LANGUAGE plpgsql; diff --git a/backend/src/CodeSpace.Core/Persistence/Entities/AgentRunLogStream.cs b/backend/src/CodeSpace.Core/Persistence/Entities/AgentRunLogStream.cs index b443feb9b..00150628e 100644 --- a/backend/src/CodeSpace.Core/Persistence/Entities/AgentRunLogStream.cs +++ b/backend/src/CodeSpace.Core/Persistence/Entities/AgentRunLogStream.cs @@ -43,6 +43,20 @@ public sealed class AgentRunLogStream : IEntity public DateTimeOffset? RemoteStallSince { get; set; } /// The typed refusal being waited out, in the capture bridge's error-code vocabulary. Never a second reason vocabulary, and never parsed. public string? RemoteStallCode { get; set; } + /// + /// The earliest instant this stream's bytes may be reclaimed, written by the retention reaper on the FIRST sweep + /// that found the stream terminal, past its rule and cited by nobody. Null means no sweep has ever proposed it — + /// which is what every stream a live worker is still capturing reads, and what the whole table read before the + /// plane existed. + /// + public DateTimeOffset? RetainUntil { get; set; } + + /// + /// When this stream's segment bytes were reclaimed. The head row deliberately outlives its bytes: a reader that + /// finds nothing cannot tell "purged by policy" from "lost", so the tombstone is what lets the Room say purged. + /// + public DateTimeOffset? PurgedAt { get; set; } + public ArtifactDigestAlgorithm? ContentDigestAlgorithm { get; set; } public byte[]? ContentDigest { get; set; } public byte[]? ManifestDigest { get; set; } diff --git a/backend/src/CodeSpace.Core/Persistence/Entities/PairedQualificationResultPin.cs b/backend/src/CodeSpace.Core/Persistence/Entities/PairedQualificationResultPin.cs new file mode 100644 index 000000000..257becae0 --- /dev/null +++ b/backend/src/CodeSpace.Core/Persistence/Entities/PairedQualificationResultPin.cs @@ -0,0 +1,46 @@ +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Core.Persistence.Entities; + +/// +/// One durable record a sealed cites, written in the seal's own transaction. +/// The retention planes read these rows as the proof that a record is still referenced — a result is a statement ABOUT +/// a set of runs, and nothing else in the schema expresses that. +/// +/// is a separate column rather than a typed use of because +/// ArtifactReferenceOracle enumerates reference sites by COLUMN NAME: an artifact pin has to sit in a column +/// ending in artifact_id, and nothing else may. Exactly one of the two is set, and the database's CHECK ties +/// which one to . +/// +public sealed class PairedQualificationResultPin : IEntity +{ + public Guid Id { get; set; } + public Guid ResultId { get; set; } + public DurablePinKind Kind { get; set; } + public Guid? PinnedId { get; set; } + public Guid? PinnedArtifactId { get; set; } + public DateTimeOffset PinnedAt { get; set; } + + /// The pinned record's id, whichever column carries it. + public Guid Target => PinnedId ?? PinnedArtifactId ?? Guid.Empty; + + public static PairedQualificationResultPin For(Guid resultId, DurablePinKind kind, Guid target, DateTimeOffset at) => new() + { + Id = Guid.NewGuid(), ResultId = resultId, Kind = kind, PinnedAt = at, + PinnedId = kind == DurablePinKind.Artifact ? null : target, + PinnedArtifactId = kind == DurablePinKind.Artifact ? target : null, + }; +} + +/// +/// What kind of record a pin names. Deliberately NOT : that enum is the set of classes +/// that have a retention RULE, and these are the kinds a qualification result can cite — an agent run has no rule at +/// all (it is identity, never collected) and is pinned anyway, because the other three are reached through it. +/// +public enum DurablePinKind +{ + LogStream = 1, + Artifact = 2, + CleanupReceipt = 3, + AgentRun = 4, +} diff --git a/backend/src/CodeSpace.Core/Persistence/EntityConfigurations/PairedQualificationResultPinConfiguration.cs b/backend/src/CodeSpace.Core/Persistence/EntityConfigurations/PairedQualificationResultPinConfiguration.cs new file mode 100644 index 000000000..6debc51a6 --- /dev/null +++ b/backend/src/CodeSpace.Core/Persistence/EntityConfigurations/PairedQualificationResultPinConfiguration.cs @@ -0,0 +1,27 @@ +using CodeSpace.Core.Persistence.Entities; +using Microsoft.EntityFrameworkCore; +using Microsoft.EntityFrameworkCore.Metadata.Builders; + +namespace CodeSpace.Core.Persistence.EntityConfigurations; + +/// Column names come from the global UseSnakeCaseNamingConvention(); the two CHECKs mirror migration 0235 exactly, and the unique index is an expression there (over both nullable target columns) so it is declared in SQL only. +public sealed class PairedQualificationResultPinConfiguration : IEntityTypeConfiguration +{ + public void Configure(EntityTypeBuilder builder) + { + builder.ToTable("paired_qualification_result_pin", table => + { + table.HasCheckConstraint("ck_paired_qualification_result_pin_kind", "kind IN ('LogStream', 'Artifact', 'CleanupReceipt', 'AgentRun')"); + table.HasCheckConstraint("ck_paired_qualification_result_pin_target", "num_nonnulls(pinned_id, pinned_artifact_id) = 1 AND (kind = 'Artifact') = (pinned_artifact_id IS NOT NULL)"); + }); + builder.HasKey(pin => pin.Id); + builder.Property(pin => pin.Id).ValueGeneratedNever(); + builder.Property(pin => pin.Kind).HasConversion().HasMaxLength(32); + builder.Ignore(pin => pin.Target); + + builder.HasOne().WithMany() + .HasForeignKey(pin => pin.ResultId) + .HasPrincipalKey(result => result.ObservationGroupId) + .OnDelete(DeleteBehavior.Cascade); + } +} diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/AgentRunLogService.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/AgentRunLogService.cs index 8762128a9..4ade8621d 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/AgentRunLogService.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/AgentRunLogService.cs @@ -315,6 +315,10 @@ public async Task ReadRangeAsync(AgentRunLogRangeRequest var requestedEndUnbounded = request.OffsetBytes + request.Length; var snapshot = await ReadSnapshotAsync(request.TeamId, request.StreamId, cancellationToken, request.OffsetBytes, requestedEndUnbounded).ConfigureAwait(false); if (snapshot == null) return RejectRange(AgentRunLogProblemCode.Missing); + // Before the segment walk, and before any range arithmetic: a purged stream's locations are gone, so every + // path below it would end at ArtifactMissing — the code that means bytes vanished when they should not have. + // The reader has to be told which of the two happened, so the tombstone answers first and in its own words. + if (snapshot.Metadata.PurgedAt != null) return RejectRange(AgentRunLogProblemCode.Purged, metadata: snapshot.Metadata); if (request.OffsetBytes > snapshot.Metadata.TotalBytes) return RejectRange(AgentRunLogProblemCode.InvalidRequest, metadata: snapshot.Metadata); var requestedEnd = Math.Min(snapshot.Metadata.TotalBytes, checked(request.OffsetBytes + request.Length)); @@ -623,7 +627,14 @@ private static MemoryStream ReadOnlyStream(ReadOnlyMemory bytes) => private static AgentRunLogOpenResult.Opened Opened(AgentRunLogStream value, bool alreadyOpen, bool reclaimed) => new(Project(value), alreadyOpen, reclaimed) { CaptureSourceBaseOffsetBytes = value.CaptureSourceBaseOffsetBytes, CaptureFinalizedAt = value.CaptureFinalizedAt }; private static AgentRunLogCaptureHead CaptureHead(AgentRunLogStream value) => new(Project(value), value.WorkerFenceEpoch!.Value, value.CaptureSessionId!.Value, value.CaptureSourceBaseOffsetBytes, value.CaptureFinalizedAt); private static AgentRunLogSegmentReceipt Receipt(AgentRunLogSegment value) => new(value.Id, value.SegmentOrdinal, value.StartOffsetBytes, value.LengthBytes, value.SourceStartOffsetBytes, value.SourceLengthBytes, value.ArtifactObjectId); - private static AgentRunLogMetadata Project(AgentRunLogStream value) => new(value.Id, value.AgentRunId, value.StreamKind, value.ContentType, value.ContentEncoding, value.CaptureSource, value.Retention, value.State, value.Revision, value.SegmentCount, value.TotalBytes, value.SourceOffsetBytes, value.ContentDigest == null ? null : Convert.ToHexStringLower(value.ContentDigest), value.CreatedAt, value.LastModifiedAt, value.CompletedAt, value.ErrorCode) { Integrity = ProjectIntegrity(value.SchemaVersion, value.ManifestDigest, value.SegmentCount, value.TotalBytes, value.CompletedAt) }; + private static AgentRunLogMetadata Project(AgentRunLogStream value) => new(value.Id, value.AgentRunId, value.StreamKind, value.ContentType, value.ContentEncoding, value.CaptureSource, value.Retention, value.State, value.Revision, value.SegmentCount, value.TotalBytes, value.SourceOffsetBytes, value.ContentDigest == null ? null : Convert.ToHexStringLower(value.ContentDigest), value.CreatedAt, value.LastModifiedAt, value.CompletedAt, value.ErrorCode) + { + // A purged stream keeps its manifest receipt, and projecting it as an integrity proof would state that these + // bytes were verified — about bytes that are gone. The receipt stays in the row as the record of what WAS + // verified; the projection withholds it, and PurgedAt is what the reader gets instead. + Integrity = value.PurgedAt == null ? ProjectIntegrity(value.SchemaVersion, value.ManifestDigest, value.SegmentCount, value.TotalBytes, value.CompletedAt) : null, + PurgedAt = value.PurgedAt, + }; private static bool SameIdentity(AgentRunLogStream stream, AgentRunLogOpenRequest request) => stream.ContentType == request.ContentType && stream.ContentEncoding == request.ContentEncoding && stream.CaptureSource == request.CaptureSource && stream.Retention == request.Retention && stream.ExpiresAt == request.ExpiresAt; private static bool Valid(AgentRunLogOpenRequest value, DateTimeOffset now) => value.TeamId != Guid.Empty && value.AgentRunId != Guid.Empty && value.WorkerFenceEpoch > 0 && value.CaptureSessionId != Guid.Empty && KeyPattern().IsMatch(value.StreamKind ?? "") && KeyPattern().IsMatch(value.CaptureSource ?? "") && value.ContentType is { Length: <= 255 } && ContentTypePattern().IsMatch(value.ContentType) && (value.ContentEncoding == null || EncodingPattern().IsMatch(value.ContentEncoding)) && Enum.IsDefined(value.Retention) && (value.ExpiresAt == null || value.ExpiresAt > now) && (value.Retention != ArtifactRetention.Ephemeral || value.ExpiresAt != null) && (value.Retention != ArtifactRetention.Permanent || value.ExpiresAt == null); private static bool Valid(AgentRunLogAppendRequest value) => value.TeamId != Guid.Empty && value.AgentRunId != Guid.Empty && value.StreamId != Guid.Empty && value.WorkerFenceEpoch > 0 && value.CaptureSessionId != Guid.Empty && value.ExpectedSegmentOrdinal > 0 && value.ExpectedOffsetBytes >= 0 && value.ExpectedSourceOffsetBytes >= 0 && value.SourceLengthBytes > 0 && value.StorageProfileId != Guid.Empty && value.StorageProfileRevision > 0 && value.ActorId != Guid.Empty && value.Bytes.Length is > 0 and <= MaximumAppendBytes && ValidTimeout(value.OperationTimeout); diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/IAgentRunLogService.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/IAgentRunLogService.cs index 84191d702..431f27486 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/IAgentRunLogService.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunLogging/IAgentRunLogService.cs @@ -142,6 +142,14 @@ public sealed record AgentRunLogMetadata( string? ErrorCode) { public AgentRunLogIntegrity? Integrity { get; init; } + + /// + /// When the retention plane reclaimed this stream's bytes, or null while they are still at their destination. It + /// rides on the metadata rather than only on the read result because a caller that LISTS streams never attempts a + /// read: without it, GET /logs would report a reclaimed archive as a completed capture with its byte count + /// intact, which is the precise lie the tombstone exists to prevent. + /// + public DateTimeOffset? PurgedAt { get; init; } } public sealed record AgentRunLogSegmentReceipt(Guid SegmentId, long SegmentOrdinal, long StartOffsetBytes, long LengthBytes, long SourceStartOffsetBytes, long SourceLengthBytes, Guid ArtifactObjectId); @@ -246,4 +254,11 @@ public enum AgentRunLogProblemCode StorageActivationFailed, ProviderTimeout, Unsupported, + + /// + /// The stream settled and its bytes were later reclaimed by the retention plane. Deliberately NOT + /// : that code says an object which should be there is not, and folding a policy + /// reclamation into it would make a healthy deployment indistinguishable from one losing data. + /// + Purged, } diff --git a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs index b2af81bef..c3bc4a6ce 100644 --- a/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs +++ b/backend/src/CodeSpace.Core/Services/Agents/Eval/Benchmark/PairedQualificationResultStore.cs @@ -58,10 +58,45 @@ public async Task SealAsync(PairedQualificationSealR ExpectedObservationCount = ExpectedCount(protocol, request.Manifest), ObservationCount = observations.Count, QualifiedForCapabilityClaim = outcome.QualifiedForCapabilityClaim, OutcomeJson = outcomeJson, }); + await PinCitedRecordsAsync(protocol, observations, cancellationToken).ConfigureAwait(false); await _db.SaveChangesAsync(cancellationToken).ConfigureAwait(false); return sealedOutcome with { ResultDigest = resultDigest }; } + /// + /// Records which durable records this result cites, staged into the SAME SaveChanges as the result row. + /// That is the whole guarantee: a seal that commits has its pins, and a seal that does not leaves none — there is + /// no window in which a sealed result exists whose evidence a reaper is free to reclaim. + /// + /// The closure is column-derived, never inferred: the observations name their agent runs, and a run reaches + /// its log streams, its cleanup receipts and the offloaded payloads of its events. Migration 0235 backfills the + /// same four closures for results sealed before this existed. + /// + private async Task PinCitedRecordsAsync(PairedQualificationProtocol protocol, IReadOnlyList observations, CancellationToken cancellationToken) + { + var resultId = protocol.ObservationGroupId; + var runIds = observations.Where(row => row.AgentRunId.HasValue).Select(row => row.AgentRunId!.Value).Distinct().ToList(); + + if (runIds.Count == 0) return; + + // Team-scoped wherever the row carries a team: the run ids come from observations this protocol's own + // validation already bound to its team, and a lookup that ignored the tenant would pin another team's rows + // against this result — which would make THEIR records unreclaimable on the strength of a seal they never saw. + // agent_run_event carries no team column of its own (its parent run does), so it is scoped by run alone. + var now = DateTimeOffset.UtcNow; + var streamIds = await _db.AgentRunLogStream.AsNoTracking().Where(stream => stream.TeamId == protocol.TeamId && runIds.Contains(stream.AgentRunId)).Select(stream => stream.Id).ToListAsync(cancellationToken).ConfigureAwait(false); + var receiptIds = await _db.AgentRunCleanupReceipt.AsNoTracking().Where(receipt => receipt.TeamId == protocol.TeamId && runIds.Contains(receipt.AgentRunId)).Select(receipt => receipt.Id).ToListAsync(cancellationToken).ConfigureAwait(false); + var artifactIds = await _db.AgentRunEvent.AsNoTracking().Where(row => runIds.Contains(row.AgentRunId) && row.DataArtifactId != null).Select(row => row.DataArtifactId!.Value).Distinct().ToListAsync(cancellationToken).ConfigureAwait(false); + + Pin(DurablePinKind.AgentRun, runIds); + Pin(DurablePinKind.LogStream, streamIds); + Pin(DurablePinKind.CleanupReceipt, receiptIds); + Pin(DurablePinKind.Artifact, artifactIds); + + void Pin(DurablePinKind kind, IEnumerable targets) => + _db.PairedQualificationResultPin.AddRange(targets.Select(target => PairedQualificationResultPin.For(resultId, kind, target, now))); + } + /// Verify the frozen runtime before the terminal row commits, so a seal minted on a substituted runtime leaves no result at all. Exempt for a replay — see . private async Task EnsureRuntimeUnchangedAsync(PairedQualificationSealRequest request, CancellationToken cancellationToken) { diff --git a/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs b/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs index 67dd4814a..a94c9dbee 100644 --- a/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs +++ b/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomNarrative.cs @@ -576,8 +576,9 @@ private static string SpendDetail(decimal? usd, int unknown, string noun) RoomAgentLogStatus.Incomplete => 0, RoomAgentLogStatus.Stalled => 1, RoomAgentLogStatus.Finalizing => 2, - RoomAgentLogStatus.Captured => 3, - _ => 4, + RoomAgentLogStatus.Purged => 3, + RoomAgentLogStatus.Captured => 4, + _ => 5, }; private static string LogStatusWord(RoomAgentLogStatus status) => status switch @@ -585,6 +586,7 @@ private static string SpendDetail(decimal? usd, int unknown, string noun) RoomAgentLogStatus.Incomplete => "incomplete", RoomAgentLogStatus.Stalled => "held; storage unavailable", RoomAgentLogStatus.Finalizing => "finalizing", + RoomAgentLogStatus.Purged => "purged; retention window elapsed", RoomAgentLogStatus.Captured => "captured; integrity proof unavailable", _ => "integrity verified", }; @@ -596,6 +598,9 @@ private static string SpendDetail(decimal? usd, int unknown, string noun) // vocabulary for it would change every renderer for a state that is not a failure. RoomAgentLogStatus.Stalled => NarrativeTone.Info, RoomAgentLogStatus.Finalizing => NarrativeTone.Info, + // Info, not Error: nothing failed. The capture settled and the bytes were reclaimed on schedule; the WORD is + // what tells the reader they are gone. + RoomAgentLogStatus.Purged => NarrativeTone.Info, RoomAgentLogStatus.Captured => NarrativeTone.Info, _ => NarrativeTone.Success, }; diff --git a/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomProjector.cs b/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomProjector.cs index 75f4c2513..723574d32 100644 --- a/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomProjector.cs +++ b/backend/src/CodeSpace.Core/Services/Sessions/Room/RoomProjector.cs @@ -218,7 +218,7 @@ private async Task> TerminalEvidence var logRows = await (from stream in _db.AgentRunLogStream.AsNoTracking() join agent in _db.AgentRun.AsNoTracking() on new { stream.TeamId, AgentRunId = stream.AgentRunId } equals new { agent.TeamId, AgentRunId = agent.Id } where stream.TeamId == teamId && agent.WorkflowRunId.HasValue && runIds.Contains(agent.WorkflowRunId.Value) - select new RunAgentLogRow(agent.WorkflowRunId.GetValueOrDefault(), stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null)) + select new RunAgentLogRow(agent.WorkflowRunId.GetValueOrDefault(), stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null, stream.PurgedAt != null)) .ToListAsync(cancellationToken).ConfigureAwait(false); var reservations = await _db.BudgetReservation.AsNoTracking() .Where(row => row.TeamId == teamId && runIds.Contains(row.WorkflowRunId)) @@ -743,7 +743,7 @@ private async Task> AgentLogsAsyn var rows = await _db.AgentRunLogStream.AsNoTracking() .Where(stream => stream.TeamId == teamId && agentIds.Contains(stream.AgentRunId)) - .Select(stream => new AgentLogRow(stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null)) + .Select(stream => new AgentLogRow(stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null, stream.PurgedAt != null)) .ToListAsync(cancellationToken).ConfigureAwait(false); return rows.GroupBy(row => row.AgentRunId).ToDictionary(group => group.Key, group => SummarizeLogs(group.ToList())); @@ -757,12 +757,19 @@ internal static RoomAgentLogSummary SummarizeLogs(IReadOnlyList row // bytes are queued in the sandbox spool. Saying "finalizing" through a storage incident is the reading an // operator acts on wrongly — it claims progress that is not happening and hides the one fact worth knowing. var stalled = rows.Count(row => row.State == AgentRunLogStreamState.Open && row.RemoteStalled); - var verified = rows.Count(row => row.State == AgentRunLogStreamState.Completed && row.SchemaVersion == 3 && row.HasManifestDigest); - var captured = rows.Count(row => row.State == AgentRunLogStreamState.Completed) - verified; - var status = incomplete.Count > 0 ? RoomAgentLogStatus.Incomplete : stalled > 0 ? RoomAgentLogStatus.Stalled : open > 0 ? RoomAgentLogStatus.Finalizing : captured > 0 ? RoomAgentLogStatus.Captured : RoomAgentLogStatus.Verified; + // A stream the retention plane reclaimed is counted apart from — never inside — captured or verified. Its + // manifest receipt is still on the row, so folding it as "integrity verified" would answer the operator's + // actual question ("can I read this?") with a proof about bytes that are gone. + var purged = rows.Count(row => row.State == AgentRunLogStreamState.Completed && row.Purged); + var settled = rows.Where(row => row.State == AgentRunLogStreamState.Completed && !row.Purged).ToList(); + var verified = settled.Count(row => row.SchemaVersion == 3 && row.HasManifestDigest); + var captured = settled.Count - verified; + var status = incomplete.Count > 0 ? RoomAgentLogStatus.Incomplete : stalled > 0 ? RoomAgentLogStatus.Stalled : open > 0 ? RoomAgentLogStatus.Finalizing + : purged > 0 ? RoomAgentLogStatus.Purged : captured > 0 ? RoomAgentLogStatus.Captured : RoomAgentLogStatus.Verified; var details = incomplete.GroupBy(row => row.State).OrderBy(group => LogStateRank(group.Key)).Select(group => $"{group.Count()} {LogStateWord(group.Key)}").ToList(); if (stalled > 0) details.Add($"{stalled} held; storage unavailable"); if (open - stalled > 0) details.Add($"{open - stalled} finalizing"); + if (purged > 0) details.Add($"{purged} purged; retention window elapsed"); if (captured > 0) details.Add($"{captured} captured; integrity proof unavailable"); if (verified > 0) details.Add($"{verified} integrity verified"); @@ -839,7 +846,7 @@ private static string DescribeRecovery(int orphaned, IReadOnlyList hosts private static readonly IReadOnlyDictionary EmptyTerminalEvidence = new Dictionary(); private static readonly IReadOnlyList EmptyAgentRows = Array.Empty(); - internal readonly record struct AgentLogRow(Guid AgentRunId, AgentRunLogStreamState State, int SchemaVersion, bool HasManifestDigest, bool RemoteStalled = false); + internal readonly record struct AgentLogRow(Guid AgentRunId, AgentRunLogStreamState State, int SchemaVersion, bool HasManifestDigest, bool RemoteStalled = false, bool Purged = false); /// One agent run of the turn, in the three columns a produced artifact has to be able to speak for. Internal so the producer fold is unit-pinned directly, not only through a full projection. internal readonly record struct AgentProducerRow(Guid AgentRunId, Messages.Enums.AgentRunStatus Status, string? ConfinementJson); @@ -847,9 +854,9 @@ private static string DescribeRecovery(int orphaned, IReadOnlyList hosts /// The per-UNIT facts every delivered artifact attaches — each unit's graded result, its log fold, and its producer record. Gathered once and handed to BOTH the deliveries and the deliverables projection, so a repository and a file can never attribute the same agent differently. private sealed record UnitTruth(IReadOnlyList Results, IReadOnlyDictionary Logs, IReadOnlyDictionary Producers); - private readonly record struct RunAgentLogRow(Guid RunId, Guid AgentRunId, AgentRunLogStreamState State, int SchemaVersion, bool HasManifestDigest, bool RemoteStalled) + private readonly record struct RunAgentLogRow(Guid RunId, Guid AgentRunId, AgentRunLogStreamState State, int SchemaVersion, bool HasManifestDigest, bool RemoteStalled, bool Purged) { - public AgentLogRow Log => new(AgentRunId, State, SchemaVersion, HasManifestDigest, RemoteStalled); + public AgentLogRow Log => new(AgentRunId, State, SchemaVersion, HasManifestDigest, RemoteStalled, Purged); } /// One agent row of a COLLAPSED turn, carrying its owning run so the batched read can be split per turn. Its is the identical row a fresh projection folds. @@ -1319,8 +1326,13 @@ private static RoomOracleProtection ProtectionOf(string? detail) return RoomOracleProtection.None; } - /// A stream that settled (Captured or Verified) reads complete; still Finalizing or Incomplete does not. - private static bool LogsAreComplete(RoomAgentLogStatus status) => status is RoomAgentLogStatus.Verified or RoomAgentLogStatus.Captured; + /// + /// A stream that settled (Captured, Verified, or settled and since Purged) reads complete; still Finalizing or + /// Incomplete does not. Purged belongs here because this field answers whether the capture SETTLED, not whether + /// the bytes are still on a destination — reading a reclaimed archive as an incomplete capture would retro-brand + /// a clean run as a broken one every time a retention window elapsed. The status word carries the bytes. + /// + private static bool LogsAreComplete(RoomAgentLogStatus status) => status is RoomAgentLogStatus.Verified or RoomAgentLogStatus.Captured or RoomAgentLogStatus.Purged; private static string? ClipVerificationDetail(string? detail) => string.IsNullOrEmpty(detail) ? null : detail.Length <= MaxVerificationDetailChars ? detail : detail[..MaxVerificationDetailChars].TrimEnd() + "…"; diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs index 8010dfc08..13619d4da 100644 --- a/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs +++ b/backend/src/CodeSpace.Core/Services/Workflows/Artifacts/Retention/ArtifactReferenceOracle.cs @@ -34,6 +34,7 @@ public sealed class ArtifactReferenceOracle : IArtifactReferenceOracle, IScopedD ("workflow_run_tool_call_attempt", "result_artifact_id"), ("workflow_run_tool_call_attempt", "error_artifact_id"), ("workflow_run_sensitive_record_payload", "ciphertext_artifact_id"), + ("paired_qualification_result_pin", "pinned_artifact_id"), ]; private static readonly string ExistsSql = BuildExistsSql(); diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs new file mode 100644 index 000000000..510d36c9e --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/Cursors/LogStreamRetentionCursor.cs @@ -0,0 +1,368 @@ +using CodeSpace.Core.DependencyInjection; +using CodeSpace.Core.Persistence.Db; +using CodeSpace.Core.Persistence.Entities; +using CodeSpace.Core.Services.Workflows.Artifacts.Runtime; +using CodeSpace.Messages.Artifacts; +using CodeSpace.Messages.Constants; +using CodeSpace.Messages.Retention; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Logging; + +namespace CodeSpace.Core.Services.Workflows.Retention.Cursors; + +/// +/// Reclaims the routed CAS bytes behind a terminal agent-run log stream that nothing cites any more. +/// +/// The bytes go, the head row stays. A purge removes the segment objects' bytes and then stamps +/// purged_at on the stream head; the head, its segments and its verification manifest survive. That is +/// deliberate: a reader that finds no stream at all cannot tell "reclaimed by policy" from "capture lost it", and the +/// entire point of a retention plane is that reclamation is legible. The read path answers a purged stream with its +/// own typed code and the Room says purged. +/// +/// Order, and what a crash leaves. The bytes go first and the tombstone commits last. A crash in between +/// leaves bytes gone and no tombstone — the next sweep re-asks, finds every location already Purged, and finishes the +/// stamp; the drain is idempotent by construction. The opposite order would leave bytes that no surviving row +/// remembers how to reach, which is the one outcome nothing can repair. +/// +/// What is never collected. Anything still names. And anything at all when +/// the question could not be asked: every failure answers , never +/// "unreferenced". +/// +/// Placements, not objects. Two byte-identical writes deduplicate to ONE content-addressed object with a +/// placement per write — and, because the object key is chosen per write, both can sit at the SAME destination. The +/// drain therefore walks placements and NAMES each one in its purge request: an unnamed claim refuses an object with +/// more than one placement outright (the coordinator will not guess which), which would make a stream that captured +/// the same bytes twice permanently unreclaimable. +/// +/// A placement the destination will not give up — a Corrupt one, which is claimable but never +/// deletable, or a destination that is refusing — is a keep, logged with the refusal's own code on every sweep and +/// deferred for the class's recheck interval. Deliberately not a terminal state: this plane has nowhere to record +/// one, and a refusal that becomes silent is how bytes stop being accounted for. +/// +public sealed class LogStreamRetentionCursor : IDurableRetentionCursor, IScopedDependency +{ + /// + /// Placements purged per stream per sweep. A 4 GiB capture is thousands of segments, and a sweep that tried to drain + /// one in a single pass would hold a worker for minutes on provider I/O. A partial drain deliberately writes + /// NOTHING, so the stream is re-claimed on the very next tick and continues where it stopped — unlike a refusal, + /// which defers it for the class's recheck interval. + /// + internal const int MaxPlacementsPerSweep = 64; + + /// + /// Every place a log stream's CAS object can still be named, as (what, why). This list IS the safety of the + /// purge, exactly as ArtifactReferenceOracle.ReferenceSites is for artifacts: a citer missing from it makes + /// the cursor answer "unreferenced" about bytes something still reaches, which is the one failure mode of this + /// class that destroys data. Public and pinned by a test, so a new citer is a deliberate edit. + /// + public static readonly IReadOnlyList<(string Table, string Column)> CitationSites = + [ + ("paired_qualification_result_pin", "pinned_id"), + ("agent_run_log_segment", "artifact_object_id"), + ("workflow_artifact", "cas_artifact_object_id"), + ]; + + /// + /// The one column that names an artifact object and is deliberately NOT a citation site, recorded here because + /// leaving it out silently is how a list like this rots. artifact_transfer_intent.artifact_object_id is + /// null on every non-terminal row — ck_artifact_transfer_intent_outcome (migration 0127) requires it — so + /// a saga in flight cannot name the object it is moving, and every row that DOES name one is a finished transfer + /// whose record is history. Probing it would answer "still cited" about every log stream in the deployment for + /// ever, because a transfer is exactly what wrote each segment, and it would look like the cursor working. + /// + public const string CommittedTransfersAreNotCiters = "artifact_transfer_intent.artifact_object_id"; + + private readonly DbContextOptions _dbOptions; + private readonly IArtifactCasPurgeCoordinator _purge; + private readonly ILogger _logger; + + public LogStreamRetentionCursor(DbContextOptions dbOptions, IArtifactCasPurgeCoordinator purge, ILogger logger) + { + _dbOptions = dbOptions; + _purge = purge; + _logger = logger; + } + + public DurableRecordClass Class => DurableRecordClass.LogStream; + + /// + /// The per-team head of the eligible queue. DISTINCT ON (team_id) is materialized first so one tenant with + /// a long backlog cannot starve the others, and the same guards are repeated on the outer select because quals + /// inside a materialized CTE are not re-evaluated against a row another worker settled in between. + /// + public async Task> ClaimAsync(DurableRetentionSweepWindow window, int limit, CancellationToken cancellationToken) + { + ArgumentNullException.ThrowIfNull(window); + await using var db = CreateDb(); + + var rows = await db.AgentRunLogStream.FromSqlInterpolated($$""" + WITH fair AS MATERIALIZED ( + SELECT DISTINCT ON (stream.team_id) stream.id, stream.completed_at + FROM agent_run_log_stream stream + WHERE stream.state <> 'Open' AND stream.purged_at IS NULL + AND stream.completed_at IS NOT NULL AND stream.completed_at <= {{window.TerminalBefore}} + AND stream.last_modified_at <= {{window.RecheckBefore}} + ORDER BY stream.team_id, stream.completed_at, stream.id + ) + SELECT stream.*, stream.xmin FROM agent_run_log_stream stream + JOIN fair ON fair.id = stream.id + WHERE stream.state <> 'Open' AND stream.purged_at IS NULL + AND stream.completed_at IS NOT NULL AND stream.completed_at <= {{window.TerminalBefore}} + AND stream.last_modified_at <= {{window.RecheckBefore}} + ORDER BY stream.completed_at, stream.id + LIMIT {{limit}} + """).AsNoTracking().ToListAsync(cancellationToken).ConfigureAwait(false); + + return rows.Select(stream => new DurableRetentionCandidate(stream.Id, stream.TeamId, stream.Revision, stream.CompletedAt!.Value, stream.RetainUntil)).ToArray(); + } + + /// Fail-closed: any failure to probe a citation site answers indeterminate, which every consumer reads as keep. + public async Task ClassifyAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken) + { + ArgumentNullException.ThrowIfNull(candidate); + + try + { + await using var db = CreateDb(); + + if (await IsPinnedAsync(db, candidate, cancellationToken).ConfigureAwait(false)) return DurableReferenceVerdict.Referenced; + + return await SharesBytesAsync(db, candidate, cancellationToken).ConfigureAwait(false) + ? DurableReferenceVerdict.Referenced + : DurableReferenceVerdict.Unreferenced; + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) { throw; } + catch (Exception ex) + { + _logger.LogWarning(ex, "Log stream {StreamId}: citation sites could not be probed; the stream is kept", candidate.Id); + + return DurableReferenceVerdict.Indeterminate; + } + } + + public async Task SettleAsync(DurableRetentionCandidate candidate, DurableRetentionDecision decision, CancellationToken cancellationToken) + { + ArgumentNullException.ThrowIfNull(candidate); + ArgumentNullException.ThrowIfNull(decision); + + return decision.Action == DurableRetentionAction.Collect + ? await CollectAsync(candidate, decision, cancellationToken).ConfigureAwait(false) + : await StampAsync(candidate, decision.RetainUntil, null, cancellationToken).ConfigureAwait(false); + } + + /// + /// Bytes first, then the tombstone, and only for a stream this plane can account for. + /// + /// The three non-collecting exits are deliberately different. A drain still making progress writes NOTHING, + /// so the next tick continues it. A refusal defers, so one unreachable destination cannot own a batch slot. And a + /// stream whose bytes are already gone by some OTHER path is never tombstoned: stamping it would make the Room + /// say "purged; retention window elapsed" about a loss this plane did not cause, inverting the very distinction + /// the tombstone exists to preserve. + /// + private async Task CollectAsync(DurableRetentionCandidate candidate, DurableRetentionDecision decision, CancellationToken cancellationToken) + { + var drain = await DrainAsync(candidate, cancellationToken).ConfigureAwait(false); + + if (drain == DrainOutcome.Progressing) return false; + + if (drain != DrainOutcome.Drained) + { + await DeferAsync(candidate, cancellationToken).ConfigureAwait(false); + + return false; + } + + var now = await DatabaseClockAsync(cancellationToken).ConfigureAwait(false); + var stamped = await StampAsync(candidate, decision.RetainUntil, now, cancellationToken).ConfigureAwait(false); + + if (stamped) + _logger.LogInformation("Log stream {StreamId} purged for team {TeamId}: its segment bytes were reclaimed and the head row now reads purged", candidate.Id, candidate.TeamId); + + return stamped; + } + + /// + /// Removes every object this stream's segments still name, bounded per sweep. Only + /// permits a tombstone, and it requires POSITIVE evidence that this plane's own lifecycle emptied the stream: + /// segments were captured, and every object behind them now rests at Purged. + /// + private async Task DrainAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken) + { + await using var db = CreateDb(); + var objects = Objects(db, candidate); + + if (!await objects.AnyAsync(cancellationToken).ConfigureAwait(false)) return NothingCaptured(candidate); + + var placements = await ReclaimableAsync(db, candidate, objects, cancellationToken).ConfigureAwait(false); + + if (placements.Count == 0) return await DrainedOrLostAsync(db, candidate, objects, cancellationToken).ConfigureAwait(false); + + foreach (var placement in placements.Take(MaxPlacementsPerSweep)) + { + if (!await PurgePlacementAsync(candidate, placement, cancellationToken).ConfigureAwait(false)) return DrainOutcome.Refused; + } + + return placements.Count > MaxPlacementsPerSweep ? DrainOutcome.Progressing : DrainOutcome.Drained; + } + + /// The objects this stream's segments name — the entry point for every question the drain asks. + private static IQueryable Objects(CodeSpaceDbContext db, DurableRetentionCandidate candidate) => + db.AgentRunLogSegment.AsNoTracking() + .Where(segment => segment.TeamId == candidate.TeamId && segment.StreamId == candidate.Id) + .Select(segment => segment.ArtifactObjectId) + .Distinct(); + + /// + /// The next placements whose bytes are still at a destination, one row over the batch so the caller can see + /// whether another sweep is needed. Read inside the collecting pass rather than carried from the decision, so a + /// resumed drain converges instead of re-asking about placements an earlier pass already took. + /// + private static async Task> ReclaimableAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, IQueryable objects, CancellationToken cancellationToken) + { + // Projected through an anonymous type rather than straight into the record: a positional constructor is not + // always translatable, and the exception it raises is swallowed as "could not be evaluated" — which reads + // exactly like a fail-closed keep and hides a cursor that reclaims nothing. + var rows = await db.ArtifactLocation.AsNoTracking() + .Where(location => location.TeamId == candidate.TeamId && objects.Contains(location.ArtifactObjectId) + && location.State != ArtifactLocationState.Purged && location.State != ArtifactLocationState.Deleted) + .OrderBy(location => location.ArtifactObjectId).ThenBy(location => location.Id) + .Select(location => new { LocationId = location.Id, location.ArtifactObjectId }) + .Take(MaxPlacementsPerSweep + 1) + .ToListAsync(cancellationToken).ConfigureAwait(false); + + return rows.Select(row => new SegmentPlacement(row.LocationId, row.ArtifactObjectId)).ToArray(); + } + + /// + /// Nothing is left to remove, so either this plane's own lifecycle emptied the stream or something else did. Only + /// the first may be tombstoned: an object whose locations never reached Purged is a loss this cursor did + /// not cause and must not claim. Counting the unaccounted objects over ALL of them, not the batch, is what keeps + /// a stream with more objects than one sweep can hold from being declared drained on a partial view. + /// + private async Task DrainedOrLostAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, IQueryable objects, CancellationToken cancellationToken) + { + var unaccounted = await objects.CountAsync(objectId => !db.ArtifactLocation + .Any(location => location.TeamId == candidate.TeamId && location.ArtifactObjectId == objectId && location.State == ArtifactLocationState.Purged), cancellationToken).ConfigureAwait(false); + + return unaccounted == 0 ? DrainOutcome.Drained : BytesAlreadyGone(candidate, unaccounted); + } + + private DrainOutcome NothingCaptured(DurableRetentionCandidate candidate) + { + _logger.LogInformation("Log stream {StreamId}: no segment ever landed, so there are no bytes to reclaim and no purge to record", candidate.Id); + + return DrainOutcome.BytesAlreadyGone; + } + + private DrainOutcome BytesAlreadyGone(DurableRetentionCandidate candidate, int unaccounted) + { + _logger.LogWarning("Log stream {StreamId}: {UnaccountedObjects} of its segment objects hold no reclaimable location and were never purged by this plane; the head row is NOT tombstoned, so the loss is not reported as a retention purge", + candidate.Id, unaccounted); + + return DrainOutcome.BytesAlreadyGone; + } + + /// + /// One placement, NAMED. Leaving ArtifactLocationId null means "the only one", which an object with more + /// than one placement cannot answer — the coordinator refuses rather than guessing, so an unnamed claim would + /// make every deduplicated capture unreclaimable for good. + /// + private async Task PurgePlacementAsync(DurableRetentionCandidate candidate, SegmentPlacement placement, CancellationToken cancellationToken) + { + var request = new ArtifactCasPurgeRequest + { + TeamId = candidate.TeamId, ArtifactObjectId = placement.ObjectId, + ArtifactLocationId = placement.LocationId, ActorId = SystemUsers.SeederId, + }; + var outcome = await _purge.PurgeAsync(request, cancellationToken).ConfigureAwait(false); + + switch (outcome) + { + case ArtifactCasPurgeResult.Purged: + return true; + case ArtifactCasPurgeResult.Rejected rejected: + _logger.LogWarning("Log stream {StreamId}: placement {LocationId} of object {ObjectId} was not reclaimed ('{Problem}'); the stream keeps its bytes and its head row", + candidate.Id, placement.LocationId, placement.ObjectId, rejected.Problem.Code); + return false; + default: + _logger.LogWarning("Log stream {StreamId}: placement {LocationId} returned an unrecognised purge outcome; the stream is kept", candidate.Id, placement.LocationId); + return false; + } + } + + private static async Task IsPinnedAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, CancellationToken cancellationToken) => + await db.PairedQualificationResultPin.AsNoTracking() + .AnyAsync(pin => pin.Kind == DurablePinKind.LogStream && pin.PinnedId == candidate.Id, cancellationToken).ConfigureAwait(false); + + /// + /// Whether anything else holds the same physical bytes. The CAS addresses an object by content, so two streams + /// that captured identical bytes — the ordinary case for short or empty output — are ONE object with two names, + /// and removing it for one takes the other's content too. Keeping is always safe here; collecting is not. For why + /// a transfer intent is not asked, see . + /// + private static async Task SharesBytesAsync(CodeSpaceDbContext db, DurableRetentionCandidate candidate, CancellationToken cancellationToken) + { + var objects = db.AgentRunLogSegment.AsNoTracking() + .Where(segment => segment.TeamId == candidate.TeamId && segment.StreamId == candidate.Id) + .Select(segment => segment.ArtifactObjectId); + + if (await db.AgentRunLogSegment.AsNoTracking().AnyAsync(segment => segment.StreamId != candidate.Id && objects.Contains(segment.ArtifactObjectId), cancellationToken).ConfigureAwait(false)) + return true; + + return await db.WorkflowArtifact.AsNoTracking() + .AnyAsync(artifact => artifact.CasArtifactObjectId != null && objects.Contains(artifact.CasArtifactObjectId.Value), cancellationToken).ConfigureAwait(false); + } + + /// A keep that could not be settled any other way still has to advance the row's modification time, or the claim query returns it again on the very next tick. + private async Task DeferAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken) => + await StampAsync(candidate, candidate.RetainUntil, null, cancellationToken).ConfigureAwait(false); + + /// + /// The one write this cursor makes, and the only shape migration 0236's guard admits: the two retention columns, + /// a revision that advances, nothing else. The revision is also the fence — a stream another worker moved since + /// the claim matches nothing, and that sweep simply settles nothing. + /// + private async Task StampAsync(DurableRetentionCandidate candidate, DateTimeOffset? retainUntil, DateTimeOffset? purgedAt, CancellationToken cancellationToken) + { + var now = purgedAt ?? await DatabaseClockAsync(cancellationToken).ConfigureAwait(false); + await using var db = CreateDb(); + + var updated = await db.AgentRunLogStream + .Where(stream => stream.Id == candidate.Id && stream.TeamId == candidate.TeamId && stream.Revision == candidate.Revision && stream.PurgedAt == null) + .ExecuteUpdateAsync(set => set + .SetProperty(stream => stream.RetainUntil, retainUntil) + .SetProperty(stream => stream.PurgedAt, purgedAt) + .SetProperty(stream => stream.Revision, stream => stream.Revision + 1) + .SetProperty(stream => stream.LastModifiedAt, now), cancellationToken).ConfigureAwait(false); + + return updated == 1; + } + + private CodeSpaceDbContext CreateDb() => new(_dbOptions); + + private async Task DatabaseClockAsync(CancellationToken cancellationToken) + { + await using var db = CreateDb(); + + return await db.Database.SqlQueryRaw("SELECT clock_timestamp() AS \"Value\"").SingleAsync(cancellationToken).ConfigureAwait(false); + } + + /// One placement still holding bytes for a segment object, as one read saw it. + private sealed record SegmentPlacement(Guid LocationId, Guid ObjectId); + + /// What one drain attempt established. Only permits a tombstone. + private enum DrainOutcome + { + /// Every object this stream's segments name is gone through this plane's own purge lifecycle. + Drained, + + /// Objects were removed and more remain. Write nothing: the next tick continues from here. + Progressing, + + /// The destination would not give the bytes up — a refusing provider, or a Corrupt placement that is claimable but never deletable. Defer. + Refused, + + /// Nothing was captured, or the bytes left by a path this plane did not drive. Never a purge. + BytesAlreadyGone, + } +} diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs new file mode 100644 index 000000000..8838a02bb --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionDecision.cs @@ -0,0 +1,75 @@ +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Core.Services.Workflows.Retention; + +/// What a sweep decided to do with one durable record. Every member except is a keep. +public enum DurableRetentionAction +{ + /// First observation of "nothing cites this" — record the quarantine deadline, remove nothing. + Quarantine, + + /// Both waits have elapsed and nothing cites the record. The ONLY action that removes anything. + Collect, + + /// Something cites the record. Keep, and clear any quarantine the citation invalidated. + Referenced, + + /// The status cannot be established. Keep. + Indeterminate, + + /// A scheduled wait (an age floor or a quarantine window) that has not elapsed. Keep. + Wait, +} + +/// Everything one decision is allowed to depend on. A record so the decision stays a two-argument pure function. +public sealed record DurableRetentionObservation(DateTimeOffset TerminalAt, DateTimeOffset? RetainUntil, DurableReferenceVerdict Verdict, DateTimeOffset Now); + +/// +/// The ONE place that decides whether a durable record may be reclaimed. Pure, so every safety property is a table of +/// inputs rather than a claim about a distributed system: nothing here reaches a database, a clock or a store. +/// +/// Four properties live in this function and nowhere else. An unregistered class keeps. A record younger than +/// its class's age floor keeps, whatever its citation status. Any verdict other than a definite "nothing cites this" +/// keeps. And a first "nothing cites this" observation only ever quarantines — collection needs the quarantine window +/// to have elapsed on top of the age floor, which is a second, independent wait. +/// +/// is not advice: it is the value the record's quarantine marker must hold after the +/// settlement. A citation CLEARS it, which is the property that keeps the two waits independent — when a cited record +/// later stops being cited, its quarantine starts again from that observation rather than inheriting a deadline set +/// while something still pointed at it. +/// +public sealed record DurableRetentionDecision(DurableRetentionAction Action, string? Code, DateTimeOffset? RetainUntil) +{ + /// + /// Decide, from (null when the running policy registers no rule for the class) and + /// . Every branch except the last returns a KEEP; reaching + /// requires passing all of them. + /// + public static DurableRetentionDecision Decide(DurableRetentionRule? rule, DurableRetentionObservation observation) + { + ArgumentNullException.ThrowIfNull(observation); + + if (rule is null) return Indeterminate("retention-class-unregistered", observation.RetainUntil); + + var eligibleAt = observation.TerminalAt.Add(rule.MinimumAge); + + if (observation.Now < eligibleAt) return Wait("age-floor-open", observation.RetainUntil); + + if (observation.Verdict == DurableReferenceVerdict.Referenced) return Referenced(); + + if (observation.Verdict != DurableReferenceVerdict.Unreferenced) return Indeterminate("reference-status-indeterminate", observation.RetainUntil); + + if (observation.RetainUntil is not { } retainUntil) return Quarantine(observation.Now.Add(rule.QuarantineWindow)); + + return observation.Now >= retainUntil ? Collect(retainUntil) : Wait("quarantine-window-open", retainUntil); + } + + public static DurableRetentionDecision Quarantine(DateTimeOffset retainUntil) => new(DurableRetentionAction.Quarantine, null, retainUntil); + public static DurableRetentionDecision Collect(DateTimeOffset retainUntil) => new(DurableRetentionAction.Collect, null, retainUntil); + + /// A citation. The quarantine marker is cleared, because the observation it recorded has been contradicted. + public static DurableRetentionDecision Referenced() => new(DurableRetentionAction.Referenced, null, null); + + public static DurableRetentionDecision Indeterminate(string code, DateTimeOffset? retainUntil) => new(DurableRetentionAction.Indeterminate, code, retainUntil); + public static DurableRetentionDecision Wait(string code, DateTimeOffset? retainUntil) => new(DurableRetentionAction.Wait, code, retainUntil); +} diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs new file mode 100644 index 000000000..146c80fd4 --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionPolicy.cs @@ -0,0 +1,33 @@ +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Core.Services.Workflows.Retention; + +/// +/// The retention policy for the durable records that are not artifacts: which classes exist and how long their records +/// are kept. Values are committed here and changed by a pull request — there is no environment override, because a +/// mistyped retention window is unrecoverable data loss and a code review is the control that belongs in front of it. +/// +/// One class, because one cursor exists. Every other plane a reaper will eventually reach — cleanup receipts, +/// capture gaps, qualification evidence, budget reservations, terminal transfer intents — gets its rule in the same +/// commit as the cursor that acts on it, so this table can always be read as "what is actually reclaimed". +/// +/// An unregistered class is NOT an error a cursor can shrug off: returns null and the decision +/// settles Indeterminate, which keeps the record forever. That is what makes removing a class from this table safe. +/// +public static class DurableRetentionPolicy +{ + /// + /// Thirty days after the stream reached its terminal capture state, then a day of quarantine, and a day between + /// looks at a stream that was kept. Deliberately far longer than any run and longer than any Room a human is still + /// reading: the bytes this reclaims are an archive nobody opened, and the floor costs storage rather than evidence. + /// + public static readonly DurableRetentionRule LogStream = + new(DurableRecordClass.LogStream, TimeSpan.FromDays(30), TimeSpan.FromHours(24), TimeSpan.FromHours(24)); + + /// The committed table, public so a test can pin every literal value in it. + public static readonly IReadOnlyDictionary Rules = + new Dictionary { [LogStream.Class] = LogStream }; + + /// The rule for , or null when this build registers none — which every consumer reads as keep. + public static DurableRetentionRule? For(DurableRecordClass value) => Rules.TryGetValue(value, out var rule) ? rule : null; +} diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs new file mode 100644 index 000000000..11e4010af --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/DurableRetentionReaper.cs @@ -0,0 +1,159 @@ +using CodeSpace.Core.Persistence.Db; +using CodeSpace.Messages.Retention; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Logging; + +namespace CodeSpace.Core.Services.Workflows.Retention; + +/// +/// The generic claim / decide / act loop, shared by every retention plane that is not the artifact ledger. +/// +/// What is structural rather than checked. (1) The cutoff a cursor claims against is derived HERE from +/// the class's own age floor, so a cursor cannot widen its candidate set past the policy. (2) The decision is +/// and nothing else, so every plane inherits the same two waits. +/// (3) Only removes anything, and it is reachable only after both waits +/// elapsed with a definite "nothing cites this". (4) A candidate that throws is logged and KEPT — the sweep never +/// converts an unanswered question into a deletion. +/// +/// Why there is no lease. The artifact reaper owns a ledger row and fences it with an owner and an epoch. +/// These planes have no ledger, and inventing one would mean a column per plane. The fence instead comes from the two +/// writes a collection actually makes: the byte removal goes through the CAS location's own durable claim, and the +/// tombstone is a conditional update on the record's revision. Two workers sweeping the same record at the same time +/// therefore remove the bytes at most once, and exactly one of them writes the tombstone; the loser settles nothing +/// and the next sweep meets a row it never held. +/// +/// Why every decision is settled, including a keep. A record nothing can ever reclaim — one a sealed +/// result pins for good, one whose bytes another record also names — is claimed by the oldest-first query on every +/// tick. If a keep wrote nothing, a few hundred such records deployment-wide would fill the batch for ever and the +/// sweep would report a healthy-looking claim count while collecting nothing, never reaching the records behind +/// them. So a keep is a settlement too: it advances the record's own modification time, and the claim query leaves +/// it alone for the class's recheck interval. +/// +public sealed class DurableRetentionReaper : IDurableRetentionReaper +{ + /// Records claimed per cursor per sweep. The cadence is hourly and every rule is measured in days, so the ceiling changes only how promptly a backlog drains — never what is collected. + private const int BatchSize = 200; + + /// Records asked for per claim. A fair cursor returns the per-tenant head, so the batch above is reached by asking repeatedly rather than in one query, and a tenant with a long backlog never holds the whole batch. + private const int ClaimSize = 25; + + private static readonly TimeSpan CandidateTimeout = TimeSpan.FromMinutes(2); + + private readonly DbContextOptions _dbOptions; + private readonly IReadOnlyList _cursors; + private readonly ILogger _logger; + + public DurableRetentionReaper(DbContextOptions dbOptions, IEnumerable cursors, ILogger logger) + { + _dbOptions = dbOptions; + _cursors = cursors.OrderBy(cursor => cursor.Class).ToArray(); + _logger = logger; + } + + public async Task SweepAsync(CancellationToken cancellationToken) + { + var now = await DatabaseClockAsync(cancellationToken).ConfigureAwait(false); + var counts = new SweepCounts(); + + foreach (var cursor in _cursors) + await SweepCursorAsync(cursor, now, counts, cancellationToken).ConfigureAwait(false); + + return counts.Summary(); + } + + /// One plane's bounded pass. A class the running policy does not register claims nothing at all, which is the same "keep forever" the decision returns for it. + private async Task SweepCursorAsync(IDurableRetentionCursor cursor, DateTimeOffset now, SweepCounts counts, CancellationToken cancellationToken) + { + var rule = DurableRetentionPolicy.For(cursor.Class); + + if (rule is null) + { + _logger.LogWarning("Durable retention: no rule is registered for {RecordClass}, so its records are kept and never claimed", cursor.Class); + return; + } + + var window = new DurableRetentionSweepWindow(now, now.Subtract(rule.MinimumAge), now.Subtract(rule.RecheckInterval)); + var seen = new HashSet(); + + while (counts.Claimed < BatchSize) + { + var limit = Math.Min(ClaimSize, BatchSize - counts.Claimed); + var claimed = await cursor.ClaimAsync(window, limit, cancellationToken).ConfigureAwait(false); + // A cursor that claims fairly hands back one record per tenant, so the batch is reached by asking again — + // and the ids already settled this tick are excluded, because a settlement that writes nothing (a drain + // still making progress) leaves the record eligible and the loop would otherwise spin on it. + var candidates = claimed.Where(candidate => seen.Add(candidate.Id)).ToList(); + + if (candidates.Count == 0) break; + + counts.Claimed += candidates.Count; + + foreach (var candidate in candidates) + counts.Record(await SweepCandidateAsync(cursor, rule, candidate, now, cancellationToken).ConfigureAwait(false)); + } + } + + /// One candidate, start to finish. Every exit that is not a completed settlement keeps the record. + private async Task SweepCandidateAsync(IDurableRetentionCursor cursor, DurableRetentionRule rule, DurableRetentionCandidate candidate, + DateTimeOffset now, CancellationToken cancellationToken) + { + using var operation = CancellationTokenSource.CreateLinkedTokenSource(cancellationToken); + operation.CancelAfter(CandidateTimeout); + + try + { + var verdict = await cursor.ClassifyAsync(candidate, operation.Token).ConfigureAwait(false); + var decision = DurableRetentionDecision.Decide(rule, new DurableRetentionObservation(candidate.TerminalAt, candidate.RetainUntil, verdict, now)); + + return await ApplyAsync(cursor, candidate, decision, operation.Token).ConfigureAwait(false); + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) { throw; } + catch (Exception ex) + { + _logger.LogWarning(ex, "Durable retention: {RecordClass} {RecordId} could not be evaluated; the record is kept", cursor.Class, candidate.Id); + + return DurableRetentionAction.Indeterminate; + } + } + + /// + /// Every decision is handed to the cursor, including the keeps: a keep that wrote nothing would be re-claimed on + /// every tick for ever. A settlement the cursor could not complete — a lost race, a refused removal, a drain that + /// needs another sweep — is reported as the keep it is. + /// + private static async Task ApplyAsync(IDurableRetentionCursor cursor, DurableRetentionCandidate candidate, + DurableRetentionDecision decision, CancellationToken cancellationToken) => + await cursor.SettleAsync(candidate, decision, cancellationToken).ConfigureAwait(false) + ? decision.Action + : DurableRetentionAction.Indeterminate; + + /// The database's clock, not the worker's: every deadline these decisions compare against was written by a database clock too. + private async Task DatabaseClockAsync(CancellationToken cancellationToken) + { + await using var db = new CodeSpaceDbContext(_dbOptions); + + return await db.Database.SqlQueryRaw("SELECT clock_timestamp() AS \"Value\"").SingleAsync(cancellationToken).ConfigureAwait(false); + } + + private sealed class SweepCounts + { + public int Claimed { get; set; } + private int Quarantined { get; set; } + private int Collected { get; set; } + private int Referenced { get; set; } + private int Kept { get; set; } + + public void Record(DurableRetentionAction action) + { + if (action == DurableRetentionAction.Quarantine) Quarantined++; + else if (action == DurableRetentionAction.Collect) Collected++; + else if (action == DurableRetentionAction.Referenced) Referenced++; + else Kept++; + } + + public DurableRetentionSweepSummary Summary() => new() + { + Claimed = Claimed, Quarantined = Quarantined, Collected = Collected, Referenced = Referenced, Kept = Kept, + }; + } +} diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs new file mode 100644 index 000000000..847f002b5 --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionCursor.cs @@ -0,0 +1,67 @@ +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Core.Services.Workflows.Retention; + +/// Whether anything still cites one durable record, as far as its cursor can establish it. +public enum DurableReferenceVerdict +{ + /// Something still cites the record — a qualification pin, or another holder of the same bytes. Never collect. + Referenced = 1, + + /// Nothing cites the record at ANY enumerated site. A necessary condition for collection, never a sufficient one. + Unreferenced = 2, + + /// The question could not be answered. Read as "keep" everywhere it is consumed. + Indeterminate = 3, +} + +/// +/// One record a sweep is considering, in the only terms the generic loop needs: who it is, when its retention clock +/// started, and what an earlier sweep already decided about it. Plane-neutral on purpose — the cursor keeps whatever +/// else it needs to settle the row. is the fence: a settlement applies only while the +/// record still holds the revision the claim saw. +/// +public sealed record DurableRetentionCandidate(Guid Id, Guid TeamId, long Revision, DateTimeOffset TerminalAt, DateTimeOffset? RetainUntil); + +/// +/// The three instants one sweep measures everything against, computed once by the loop from the class's own rule so a +/// cursor cannot widen its candidate set past the policy. +/// +/// The database clock at sweep start. Every deadline compared against it was written by a database clock too. +/// The age floor: a record that went terminal after this is not a candidate at all. +/// The deferral: a record last settled after this was already looked at recently and is left alone. +public sealed record DurableRetentionSweepWindow(DateTimeOffset Now, DateTimeOffset TerminalBefore, DateTimeOffset RecheckBefore); + +/// +/// One plane's answers to the three questions the reaper asks: which records are candidates, does anything still cite +/// this one, and apply this decision. Deliberately narrow (Rule 7): no policy, no clock, no batching — those belong to +/// the loop, so every plane inherits the same waits rather than re-deriving them. +/// +/// is fail-closed by contract: any failure to reach a citation site answers +/// , never . +/// +public interface IDurableRetentionCursor +{ + DurableRecordClass Class { get; } + + /// + /// Records in a terminal state that went terminal at or before , + /// are not already collected, and were not settled since . + /// Oldest first, fairly across tenants, at most . + /// + /// The recheck half is what keeps one unreclaimable record from owning a batch slot for good: every + /// settlement — including a keep — advances the record's own modification time, and this query then leaves it + /// alone until the interval elapses. + /// + Task> ClaimAsync(DurableRetentionSweepWindow window, int limit, CancellationToken cancellationToken); + + Task ClassifyAsync(DurableRetentionCandidate candidate, CancellationToken cancellationToken); + + /// + /// Applies under the candidate's revision fence. False means nothing was settled — + /// the record moved under this sweep, or the work it needed could not be finished — and the loop reports it as a + /// keep. A cursor decides for itself whether an unfinished settlement also defers the record; work that is making + /// progress must NOT, or a drain that needs several passes would take one recheck interval per pass. + /// + Task SettleAsync(DurableRetentionCandidate candidate, DurableRetentionDecision decision, CancellationToken cancellationToken); +} diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionReaper.cs b/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionReaper.cs new file mode 100644 index 000000000..0cbf41d5b --- /dev/null +++ b/backend/src/CodeSpace.Core/Services/Workflows/Retention/IDurableRetentionReaper.cs @@ -0,0 +1,15 @@ +using CodeSpace.Core.DependencyInjection; +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Core.Services.Workflows.Retention; + +/// +/// Runs one BOUNDED retention sweep over every registered cursor. Bounded means three separate things, all of them +/// load-bearing: each cursor claims at most BatchSize records per sweep; a record is settled by its own +/// conditional write rather than a transaction spanning the batch; and a row that fails is logged and skipped, so one +/// unreachable destination cannot stop the rest of the sweep. +/// +public interface IDurableRetentionReaper : IScopedDependency +{ + Task SweepAsync(CancellationToken cancellationToken); +} diff --git a/backend/src/CodeSpace.Messages/Commands/Workflows/ReapExpiredDurableRecordsCommand.cs b/backend/src/CodeSpace.Messages/Commands/Workflows/ReapExpiredDurableRecordsCommand.cs new file mode 100644 index 000000000..a4135bf4e --- /dev/null +++ b/backend/src/CodeSpace.Messages/Commands/Workflows/ReapExpiredDurableRecordsCommand.cs @@ -0,0 +1,24 @@ +using CodeSpace.Messages.Mediation; +using CodeSpace.Messages.Retention; + +namespace CodeSpace.Messages.Commands.Workflows; + +/// +/// Run one bounded durable-retention sweep: for every registered plane, claim records past their class's age floor, +/// establish whether anything still cites each one, and reclaim only those proven uncited past both waits. +/// +/// NOT tenant-scoped — system-wide reclamation that runs without an actor context. Fired by the recurring +/// reaper job; also sendable ad-hoc from an admin path or a test. +/// +/// NOT transactional (): each record is settled by its own conditional +/// write, and a reclamation that fails leaves only that record for the next pass. One command transaction around the +/// whole tick would let a single unreachable destination undo every other record's work — and it would hold a +/// transaction open across provider I/O. +/// +public sealed record ReapExpiredDurableRecordsCommand : ICommand, INonTransactionalCommand; + +/// The sweep's per-bucket counts, surfaced for logging and for the recurring job's result. +public sealed record ReapExpiredDurableRecordsResponse +{ + public required DurableRetentionSweepSummary Summary { get; init; } +} diff --git a/backend/src/CodeSpace.Messages/Dtos/Agents/AgentRunLogDtos.cs b/backend/src/CodeSpace.Messages/Dtos/Agents/AgentRunLogDtos.cs index 356164e35..ed5b1ee56 100644 --- a/backend/src/CodeSpace.Messages/Dtos/Agents/AgentRunLogDtos.cs +++ b/backend/src/CodeSpace.Messages/Dtos/Agents/AgentRunLogDtos.cs @@ -73,6 +73,9 @@ public enum AgentRunLogReadAvailability AccessDenied, ProviderTimeout, Unsupported, + + /// The capture settled and the retention plane later reclaimed its bytes. Gone, and gone on purpose — which is exactly what does not say. + Purged, } public sealed record AgentRunLogReadProblem diff --git a/backend/src/CodeSpace.Messages/Dtos/Sessions/Room/RoomAgentLogStatus.cs b/backend/src/CodeSpace.Messages/Dtos/Sessions/Room/RoomAgentLogStatus.cs index fc6f312c9..e192d86b7 100644 --- a/backend/src/CodeSpace.Messages/Dtos/Sessions/Room/RoomAgentLogStatus.cs +++ b/backend/src/CodeSpace.Messages/Dtos/Sessions/Room/RoomAgentLogStatus.cs @@ -30,4 +30,12 @@ public enum RoomAgentLogStatus /// today, and a value shifted underneath a stored ordinal is unrecoverable. /// Stalled, + + /// + /// Settled, and its bytes have since been reclaimed by the retention plane. Distinct from every other member + /// because nothing went wrong: the capture completed, the window elapsed, and the head row survives precisely so a + /// reader is told that rather than meeting a stream whose bytes will not load. Appended for the same reason + /// was — no existing member's ordinal may move. + /// + Purged, } diff --git a/backend/src/CodeSpace.Messages/Retention/DurableRetention.cs b/backend/src/CodeSpace.Messages/Retention/DurableRetention.cs new file mode 100644 index 000000000..b05b57fec --- /dev/null +++ b/backend/src/CodeSpace.Messages/Retention/DurableRetention.cs @@ -0,0 +1,55 @@ +namespace CodeSpace.Messages.Retention; + +/// +/// A class of durable record that has a retention rule AND a cursor that can act on it. A class is added here together +/// with the cursor that sweeps it, never ahead of one: a rule with no cursor reclaims nothing and reads, to anyone +/// scanning this file, as a promise the system does not keep. +/// +/// Membership is the same test the artifact plane applies to ArtifactRetentionClass: a class exists only +/// when the COMPLETE set of places that can still cite one of its records is enumerable — a column, or a pin written in +/// the citing statement's own transaction. A record whose citations can also reach a JSON payload has no class and is +/// therefore never a reap candidate. +/// +public enum DurableRecordClass +{ + /// An agent_run_log_stream head and the routed CAS bytes its segments name. + LogStream = 1, +} + +/// +/// One class's committed rule. is the age floor measured from the record's own terminal +/// instant: below it the record is not even considered, so a citation that is still in flight cannot be outrun. +/// is the second, independent wait measured from the first observation of +/// "nothing cites this" — collection needs BOTH to have elapsed. is neither: it is +/// how long a record that was looked at and KEPT is left alone, so one record nothing can ever reclaim cannot own a +/// batch slot for good. +/// +public sealed record DurableRetentionRule(DurableRecordClass Class, TimeSpan MinimumAge, TimeSpan QuarantineWindow, TimeSpan RecheckInterval); + +/// What one bounded sweep did. Every claimed record lands in exactly one of these buckets. +public sealed record DurableRetentionSweepSummary +{ + public required int Claimed { get; init; } + + /// First observation of "nothing cites this": the quarantine deadline was recorded and nothing was removed. + public required int Quarantined { get; init; } + + /// Both waits elapsed with no citation. The only bucket that removed anything. + public required int Collected { get; init; } + + /// Something still cites the record. Kept, and not looked at again until the recheck interval elapses. + public required int Referenced { get; init; } + + /// + /// A keep this sweep could not turn into a quarantine or a collection: an unanswered citation question, a refused + /// removal, a drain that needs another pass, or a record whose bytes were already gone by some other path. Kept. + /// + /// Distinguishing this from is the point of having it: a sweep that reports Claimed + /// and nothing else is a sweep that is re-asking, and an operator reading these numbers has to be able to see + /// that rather than infer healthy work from a non-zero claim count. + /// + public required int Kept { get; init; } + + public static DurableRetentionSweepSummary Empty { get; } = + new() { Claimed = 0, Quarantined = 0, Collected = 0, Referenced = 0, Kept = 0 }; +} diff --git a/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs new file mode 100644 index 000000000..f88fa0083 --- /dev/null +++ b/backend/tests/CodeSpace.IntegrationTests/Workflows/DurableRetentionReaperFlowTests.cs @@ -0,0 +1,1076 @@ +using Autofac; +using CodeSpace.Core.Persistence.Db; +using CodeSpace.Core.Persistence.Entities; +using CodeSpace.Core.Services.Agents.AgentRunLogging; +using CodeSpace.Core.Services.Agents.Eval.Benchmark; +using CodeSpace.Core.Services.Sessions.Room; +using CodeSpace.Core.Services.Credentials; +using CodeSpace.Core.Services.Workflows.Artifacts.Credentials; +using CodeSpace.Core.Services.Workflows.Artifacts.Providers.Local; +using CodeSpace.Core.Services.Workflows.Artifacts.Retention; +using CodeSpace.Core.Services.Workflows.Artifacts.Runtime; +using CodeSpace.Core.Services.Workflows.Retention; +using CodeSpace.Core.Services.Workflows.Retention.Cursors; +using CodeSpace.IntegrationTests.Infrastructure; +using CodeSpace.Messages.Agents.Benchmark; +using CodeSpace.Messages.Artifacts; +using CodeSpace.Messages.Dtos.Sessions.Room; +using CodeSpace.Messages.Enums; +using CodeSpace.Messages.Retention; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.Logging.Abstractions; +using Npgsql; +using Shouldly; + +namespace CodeSpace.IntegrationTests.Workflows; + +/// +/// The durable-retention plane against a real database and a real local-rwx destination, where the parts that cannot +/// be unit-tested actually live: the guard trigger that decides which retention statements PostgreSQL admits, the pin +/// rows a seal writes in its own transaction, and the order in which bytes and their tombstone commit. +/// +/// The positive control at the top is what proves the counter-examples below it are not passing vacuously: each +/// of those builds a stream that is ONE property away from collectable and asserts its bytes survive. Every assertion +/// is about a row this class created — never a sweep tally, which is deployment-wide over a shared database and +/// therefore counts other classes' rows as readily as its own. +/// +/// The class also reclaims what it staged (): a bounded global sweep elsewhere in the +/// suite — the artifact-location verifier — selects the hundred least-recently-verified placements across every team, +/// so a class that leaves live placements behind a destination it then deletes changes what that sweep sees. +/// +[Collection(PostgresCollection.Name)] +[Trait("Category", "Integration")] +public sealed class DurableRetentionReaperFlowTests : IAsyncLifetime +{ + private readonly PostgresFixture _fixture; + private readonly List _roots = []; + private readonly List _staged = []; + + public DurableRetentionReaperFlowTests(PostgresFixture fixture) => _fixture = fixture; + + /// + /// The positive control, and the ordering claim with it: the bytes leave the destination FIRST and the head row's + /// tombstone commits after, so a crash between them leaves bytes gone and a row the next sweep finishes — never a + /// row that says purged about bytes still being paid for. + /// + [Fact] + public async Task A_terminal_unpinned_log_stream_past_its_rule_is_purged_blobs_first_then_rows_and_the_Room_reads_purged() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "the archived transcript of a run nobody opened again"); + await AgeAsync(stream, TimeSpan.FromDays(31)); + + await SweepAsync(); + + var quarantined = await StreamAsync(stream); + quarantined.PurgedAt.ShouldBeNull("the sweep that first noticed the stream must not touch its bytes"); + quarantined.RetainUntil.ShouldNotBeNull("the quarantine deadline has to be durable — an in-memory one is no wait at all"); + (await BytesReadableAsync(world, stream)).ShouldBeTrue(); + + await ElapseQuarantineAsync(stream); + await SweepAsync(); + + var purged = await StreamAsync(stream); + purged.PurgedAt.ShouldNotBeNull("the head row is the tombstone: it outlives its bytes so a reader is told they were reclaimed"); + purged.State.ShouldBe(AgentRunLogStreamState.Completed, "a purge is not a capture verdict — the state it settled in is untouched"); + purged.TotalBytes.ShouldBeGreaterThan(0, "the byte head still records what was captured; only the bytes themselves are gone"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(0, "every location behind the stream's segments must be drained before the tombstone commits"); + RoomProjector.SummarizeLogs([Row(purged)]).Status.ShouldBe(RoomAgentLogStatus.Purged, + "the Room must say purged rather than claim an integrity proof over bytes that are gone"); + } + + /// + /// What a reader is told about a stream whose bytes were reclaimed. A 410 alone is the answer a LOST object gets, + /// so the code beside it carries the distinction, and the metadata must stop advertising an integrity proof over + /// bytes that are gone. + /// + [Fact] + public async Task The_read_path_answers_a_purged_stream_as_purged_and_claims_no_integrity() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "bytes a reader will come looking for"); + var before = await MetadataAsync(world, stream); + before.Integrity.ShouldNotBeNull().ManifestDigest.ShouldNotBeNull("the positive control: while the bytes are there the stream does carry its proof"); + before.PurgedAt.ShouldBeNull(); + + await PurgeAsync(world, stream); + + var read = (await ReadAsync(world, stream)).ShouldBeOfType(); + read.Problem.Code.ShouldBe(AgentRunLogProblemCode.Purged, + "ArtifactMissing means an object that should be there is not; a policy reclamation must never be reported in the vocabulary of data loss"); + var after = await MetadataAsync(world, stream); + after.PurgedAt.ShouldNotBeNull("a caller that LISTS streams never attempts a read, so the tombstone has to ride on the metadata too"); + after.Integrity.ShouldBeNull("the manifest receipt is still in the row as the record of what WAS verified; projecting it now would claim these bytes are verifiable"); + after.TotalBytes.ShouldBe(before.TotalBytes, "what was captured is still stated; only the claim that it can be read is withdrawn"); + } + + /// + /// The pin-table-first property, from the other side: a stream a sealed qualification result cites is never + /// collected, however long both windows have been open. Mutation: drop the pin probe in + /// LogStreamRetentionCursor.ClassifyAsync and this test reds with the stream purged. + /// + [Fact] + public async Task A_pinned_log_stream_past_its_rule_survives() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "evidence a capability claim was computed from"); + await SealResultAsync(world); + await AgeAsync(stream, TimeSpan.FromDays(400)); + + await SweepAsync(); + + (await StreamAsync(stream)).RetainUntil.ShouldBeNull("a cited stream is never quarantined — the verdict is keep, not a wait"); + + // The strongest form of the claim: a quarantine deadline that has ALREADY elapsed, so the only thing left + // between the reaper and the bytes is the pin. + await ElapseQuarantineAsync(stream); + await SweepAsync(); + + var kept = await StreamAsync(stream); + kept.PurgedAt.ShouldBeNull("a sealed result still cites this stream; no elapsed window outranks that"); + kept.RetainUntil.ShouldBeNull("and the citation cleared the quarantine it contradicts, so an un-pinned stream starts its wait afresh"); + (await BytesReadableAsync(world, stream)).ShouldBeTrue(); + (await UnpurgedLocationsAsync(world, stream)).ShouldBeGreaterThan(0); + } + + /// + /// The starvation property. A record nothing can ever reclaim is claimed by an oldest-first query on every tick, + /// so a keep that wrote nothing would fill the batch with the same rows for ever and the sweep would report a + /// healthy claim count while collecting nothing. Mutation: stop settling the keeps (or drop the recheck predicate + /// from the claim query) and the kept stream comes back in the second claim, ahead of the eligible one. + /// + [Fact] + public async Task A_stream_that_was_kept_is_not_re_claimed_until_its_recheck_interval_elapses() + { + var world = await SeedWorldAsync(); + var pinned = await CaptureAsync(world, "evidence pinned for ever"); + await SealResultAsync(world); + await AgeAsync(pinned, TimeSpan.FromDays(400)); + await SweepAsync(); + + // A second team, because the claim is fair per tenant: two streams of ONE team can never be in the same + // batch, so a single-team version of this test could not tell starvation from that fairness. + var neighbour = await SeedWorldAsync(); + var collectable = await CaptureAsync(neighbour, "a younger archive behind it"); + await AgeAsync(collectable, TimeSpan.FromDays(31)); + var claimed = await ClaimAsync(limit: 200); + + claimed.ShouldNotContain(pinned, "a stream this sweep already looked at and kept must not be handed back on the next tick — at scale it would own the batch for ever"); + claimed.ShouldContain(collectable, "and the stream behind it, which no sweep has looked at, must be reachable"); + } + + /// + /// Throughput, which the per-tenant fairness of the claim query would otherwise cap at one record per tenant per + /// tick: with an hourly cadence and a two-visit collection that is about a dozen streams a team a day, while a + /// single run writes two. The loop asks again until the batch is full. Mutation: claim once and only the oldest + /// stream of the three is ever swept. + /// + [Fact] + public async Task One_sweep_reaches_every_eligible_stream_of_a_team_not_just_its_oldest() + { + var world = await SeedWorldAsync(); + var streams = new List(); + + foreach (var kind in new[] { AgentRunLogKinds.StandardOutput, AgentRunLogKinds.StandardError, AgentRunLogKinds.Transcript }) + { + var stream = await CaptureAsync(world, $"archive {kind}", kind); + await AgeAsync(stream, TimeSpan.FromDays(31)); + streams.Add(stream); + } + + await SweepAsync(); + + foreach (var stream in streams) + (await StreamAsync(stream)).RetainUntil.ShouldNotBeNull($"stream {stream} of the same team was never reached by the sweep"); + } + + /// + /// The atomicity claim, measured rather than asserted: xmin is the id of the transaction that inserted a + /// row, so a result and its pins sharing one xmin IS "written in the same transaction". Mutation: move the + /// pin write to its own SaveChanges (or its own scope) and the two ids differ — which is the crash window + /// in which a sealed result exists whose evidence the reaper is free to reclaim. + /// + [Fact] + public async Task A_sealed_result_pins_every_stream_and_artifact_it_cites_in_one_transaction() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "one archived stream"); + var (receiptId, artifactId) = await SeedCitedRecordsAsync(world); + + var groupId = await SealResultAsync(world); + + var pins = await PinsAsync(groupId); + pins.Where(pin => pin.Kind == DurablePinKind.AgentRun).Select(pin => pin.Target).ShouldBe([world.AgentRunId]); + pins.Where(pin => pin.Kind == DurablePinKind.LogStream).Select(pin => pin.Target).ShouldBe([stream]); + pins.Where(pin => pin.Kind == DurablePinKind.CleanupReceipt).Select(pin => pin.Target).ShouldBe([receiptId]); + pins.Where(pin => pin.Kind == DurablePinKind.Artifact).Select(pin => pin.Target).ShouldBe([artifactId]); + pins.ShouldAllBe(pin => (pin.Kind == DurablePinKind.Artifact) == (pin.PinnedArtifactId != null), + "an artifact pin has to sit in the column the artifact oracle probes by name, and nothing else may"); + + var transactions = await InsertingTransactionsAsync(groupId); + transactions.Count.ShouldBe(1, $"the result and its {pins.Count} pins must commit together; they were inserted by {transactions.Count} transactions"); + + ArtifactReferenceOracle.ReferenceSites.ShouldContain(("paired_qualification_result_pin", "pinned_artifact_id"), + "an artifact a sealed result pins has to be a probed reference site, or the ARTIFACT reaper collects what this plane is protecting"); + } + + /// + /// Migration 0235's backfill and the writer have to agree, because a result sealed before this plane existed would + /// otherwise read as citing nothing and its evidence would become collectable. Mutation: change either closure + /// (drop the cleanup receipts from the backfill, add a kind to the writer) and the two sets differ here. + /// + [Fact] + public async Task The_backfill_writes_exactly_the_pins_the_seal_writes() + { + var world = await SeedWorldAsync(); + await CaptureAsync(world, "a stream a pre-existing result cites"); + await SeedCitedRecordsAsync(world); + var groupId = await SealResultAsync(world); + var written = (await PinsAsync(groupId)).Select(Identity).ToList(); + + written.ShouldNotBeEmpty("the comparison below is vacuous unless the writer actually pinned something"); + await ForgetPinsAsync(groupId); + await RunBackfillAsync(); + + (await PinsAsync(groupId)).Select(Identity).Order(StringComparer.Ordinal) + .ShouldBe(written.Order(StringComparer.Ordinal), + "0235's backfill is the same four closures as PairedQualificationResultStore.PinCitedRecordsAsync; if they drift, results sealed before this migration lose the protection new ones get"); + } + + /// + /// Bytes before rows, proven by the failure: with the destination refusing, the sweep must leave the head row's + /// tombstone unwritten. Mutation: stamp the tombstone before (or without) the byte removal and this reds — the + /// Room would then read "purged" about an archive still sitting at the destination, and no later sweep would ever + /// reclaim it, because a purged row is never claimed again. + /// + [Fact] + public async Task A_refused_byte_purge_leaves_the_tombstone_unwritten_and_defers_the_stream() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "bytes at a destination that stops answering"); + await AgeAsync(stream, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(stream); + var beforeRefusal = await StreamAsync(stream); + + await SweepAgainstARefusingDestinationAsync(); + + var kept = await StreamAsync(stream); + kept.PurgedAt.ShouldBeNull("the tombstone must never run ahead of the bytes"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBeGreaterThan(0, "a refused delete leaves the location exactly where it was"); + kept.LastModifiedAt.ShouldBeGreaterThan(beforeRefusal.LastModifiedAt, + "a refusal is deferred rather than retried every tick, so one unreachable destination cannot own a batch slot"); + (await ClaimAsync(limit: 10)).ShouldNotContain(stream); + } + + /// + /// A drain too big for one sweep must NOT be deferred like a refusal: it is making progress, so it writes nothing + /// and the next tick continues it. Mutation: defer a partial drain and a large capture takes one recheck interval + /// per batch of objects — a 4 GiB archive would need months. + /// + [Fact] + public async Task A_drain_too_big_for_one_sweep_writes_nothing_and_the_next_sweep_finishes_it() + { + var world = await SeedWorldAsync(); + var segments = LogStreamRetentionCursor.MaxPlacementsPerSweep + 1; + var stream = await CaptureAsync(world, "first", AgentRunLogKinds.StandardOutput, segments); + await AgeAsync(stream, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(stream); + var beforeDrain = await StreamAsync(stream); + + await SweepAsync(); + + var partly = await StreamAsync(stream); + partly.PurgedAt.ShouldBeNull("the tombstone waits until every object is gone"); + partly.LastModifiedAt.ShouldBe(beforeDrain.LastModifiedAt, "a drain that is progressing writes NOTHING, so the very next tick continues it"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(segments - LogStreamRetentionCursor.MaxPlacementsPerSweep, "exactly one bounded batch of objects went"); + + await SweepAsync(); + + (await StreamAsync(stream)).PurgedAt.ShouldNotBeNull("the second sweep finishes the drain and stamps the tombstone"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(0); + } + + /// + /// Deduplication INSIDE one stream. Two byte-identical segments are one content-addressed object with a placement + /// per write, both at the same destination, and nothing else names them — so the stream is collectable and the + /// drain has to take each placement by name. Mutation: purge by object id alone and the coordinator refuses to + /// guess which placement, so a capture that repeated itself would never be reclaimed at all. + /// + [Fact] + public async Task A_stream_that_captured_the_same_bytes_twice_is_still_collected() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "identical", segments: 2, distinctSegments: false); + (await ObjectsOfAsync(world, stream)).Count.ShouldBe(1, "the premise: the CAS stored ONE object for the two identical segments"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(2, "and ONE placement per write, which an unnamed purge claim refuses to choose between"); + + await AgeAsync(stream, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(stream); + await SweepAsync(); + + (await StreamAsync(stream)).PurgedAt.ShouldNotBeNull("nothing outside this stream names these bytes, so repeating itself must not make a capture unreclaimable"); + (await UnpurgedLocationsAsync(world, stream)).ShouldBe(0, "every placement of the object goes, not just the one an unnamed claim would have found"); + } + + /// + /// Two streams that captured identical output are ONE content-addressed object under two names — the ordinary case + /// for short or empty logs. Mutation: drop the sibling-segment probe and purging either one silently empties the + /// other, whose head row still says its bytes are there. + /// + [Fact] + public async Task A_stream_whose_bytes_another_stream_also_names_is_never_collected() + { + var world = await SeedWorldAsync(); + var identical = "byte-for-byte the same output"; + var first = await CaptureAsync(world, identical); + var second = await CaptureAsync(world, identical, AgentRunLogKinds.StandardError); + (await ObjectsOfAsync(world, first)).ShouldBe(await ObjectsOfAsync(world, second), "the premise: the CAS stored one object for both captures"); + + await AgeAsync(first, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(first); + await SweepAsync(); + + (await StreamAsync(first)).PurgedAt.ShouldBeNull("another stream still reaches these bytes"); + (await UnpurgedLocationsAsync(world, first)).ShouldBeGreaterThan(0); + (await BytesReadableAsync(world, second)).ShouldBeTrue("and the stream that was never a candidate still reads"); + + // The verdict itself, because the outcome alone does not pin the reason: deduplicated captures also leave one + // object with a placement per write, which the drain refuses separately as replication. Asserting the + // CITATION verdict is what reds when the sibling-segment probe is removed. + var candidate = (await CandidatesAsync(limit: 200)).FirstOrDefault(row => row.Id == first) + ?? new DurableRetentionCandidate(first, world.TeamId, (await StreamAsync(first)).Revision, DateTimeOffset.UtcNow.AddDays(-99), null); + (await Cursor().ClassifyAsync(candidate, CancellationToken.None)).ShouldBe(DurableReferenceVerdict.Referenced, + "the stream is kept because something else NAMES its bytes, not because of how they happen to be placed"); + } + + /// + /// Why a transfer intent is NOT one of the citers the cursor probes, asserted rather than reasoned about: the + /// schema refuses to let a saga in flight name an object at all. Every row that does name one is a finished + /// transfer — and a transfer is what wrote each log segment — so probing the column would answer "still cited" + /// about every log stream in the deployment for ever while looking exactly like a safety property. + /// + [Fact] + public async Task A_transfer_saga_in_flight_cannot_name_the_object_it_is_moving() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "bytes a transfer already moved"); + var objectId = (await ObjectsOfAsync(world, stream)).Single(); + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + db.ArtifactTransferIntent.Add(await InFlightIntentAsync(db, world, objectId)); + + var raised = await Should.ThrowAsync(async () => await db.SaveChangesAsync()); + + PostgresErrorOf(raised).MessageText.ShouldContain("ck_artifact_transfer_intent_outcome", customMessage: + "if a non-terminal intent could ever carry an artifact_object_id, it WOULD be a citer and the cursor would have to probe it"); + (await db.ArtifactTransferIntent.AsNoTracking().CountAsync(intent => intent.ArtifactObjectId == objectId && intent.State != ArtifactTransferState.Committed)) + .ShouldBe(0, "the only rows naming this object are the committed transfer that wrote it"); + } + + /// + /// The age floor, at the CALL SITE rather than in the pure decision: the window the loop computes from the class's + /// rule is what keeps a young stream out of the batch entirely. Mutation: widen the window (or drop the + /// completed_at predicate) and a stream one day short of its floor is claimed. + /// + [Fact] + public async Task A_stream_inside_its_age_floor_is_never_claimed() + { + var world = await SeedWorldAsync(); + var young = await CaptureAsync(world, "a capture from last month, one day short"); + await AgeAsync(young, DurableRetentionPolicy.LogStream.MinimumAge - TimeSpan.FromDays(1)); + + (await ClaimAsync(limit: 50)).ShouldNotContain(young); + + await AgeAsync(young, TimeSpan.FromDays(2)); + + (await ClaimAsync(limit: 50)).ShouldContain(young, "the counter-example is only worth anything if the same stream IS claimed once its floor has passed"); + } + + /// + /// The claim query requires a completion instant because the age floor is measured from one. That predicate can + /// never be the ONLY thing standing between a clockless row and a purge, and this is why: the schema refuses to + /// hold a terminal stream without one at all. Asserting the refusal rather than fabricating the row keeps the + /// guarantee where it actually lives. + /// + [Fact] + public async Task A_terminal_stream_cannot_exist_without_the_instant_its_floor_is_measured_from() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "a settled capture"); + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream DISABLE TRIGGER agent_run_log_stream_enforce_invariants"); + + try + { + var raised = await Should.ThrowAsync(async () => + await db.Database.ExecuteSqlRawAsync("UPDATE agent_run_log_stream SET completed_at = NULL WHERE id = {0}", [stream])); + + PostgresErrorOf(raised).MessageText.ShouldContain("ck_agent_run_log_stream_terminal"); + } + finally + { + await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream ENABLE TRIGGER agent_run_log_stream_enforce_invariants"); + } + } + + /// + /// The fence. A settlement carries the revision its claim saw, and a stream another worker moved in between must + /// take nothing from it. Mutation: drop the revision predicate from the stamp and two workers racing one stream + /// both write, so the loser's stale verdict lands on top of the winner's. + /// + [Fact] + public async Task A_settlement_whose_revision_moved_writes_nothing() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "a stream two workers both claimed"); + await AgeAsync(stream, TimeSpan.FromDays(31)); + var claimed = (await CandidatesAsync(limit: 50)).Single(candidate => candidate.Id == stream); + await ElapseQuarantineAsync(stream); // somebody else moved the row after this claim was taken + var moved = await StreamAsync(stream); + + var settled = await Cursor().SettleAsync(claimed, DurableRetentionDecision.Quarantine(DateTimeOffset.UtcNow.AddDays(1)), CancellationToken.None); + + settled.ShouldBeFalse("the claim is stale, so it settles nothing"); + var after = await StreamAsync(stream); + after.Revision.ShouldBe(moved.Revision, "and it must leave the row exactly where the other worker put it"); + after.RetainUntil.ShouldBe(moved.RetainUntil); + } + + /// + /// The distinction the tombstone exists for. A stream whose bytes were lost by some OTHER path must not be stamped + /// "purged", or the Room would tell the operator a retention window elapsed on an archive nobody reclaimed. + /// Mutation: treat an empty reclaimable set as drained and this reds with a tombstone on a loss this plane did not + /// cause. + /// + [Fact] + public async Task A_stream_whose_bytes_vanished_outside_this_plane_is_never_tombstoned() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "bytes something else lost"); + await AgeAsync(stream, TimeSpan.FromDays(31)); + await SweepAsync(); + await ElapseQuarantineAsync(stream); + await MarkLocationsDeletedAsync(world, stream); + + await SweepAsync(); + + var kept = await StreamAsync(stream); + kept.PurgedAt.ShouldBeNull("this plane removed nothing, so it may not report a purge"); + RoomProjector.SummarizeLogs([Row(kept)]).Status.ShouldNotBe(RoomAgentLogStatus.Purged, + "\"purged; retention window elapsed\" about a loss nobody chose inverts the one distinction the tombstone preserves"); + } + + /// + /// Migration 0236's arm, at the only place that can pin it: the database. A terminal stream admits a retention + /// statement and NOTHING else — a statement that smuggles anything alongside the two columns reads the same + /// refusal it always did. + /// + [Theory] + [InlineData("retain_until = now(), state = 'Corrupt'", "terminal state is immutable")] + [InlineData("retain_until = now(), total_bytes = 0", "terminal state is immutable")] + [InlineData("retain_until = now(), completed_at = now()", "terminal state is immutable")] + [InlineData("state = 'Open', completed_at = NULL", "terminal state is immutable")] + [InlineData("purged_at = now()", "cannot be purged without the retain_until")] + [InlineData("purged_at = now(), retain_until = now() - interval '1 day'", "cannot be purged without the retain_until")] + public async Task The_guard_refuses_a_retention_statement_that_says_more_than_retention(string smuggled, string refusal) + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "a settled stream"); + using var scope = _fixture.BeginScope(); + var sql = $"UPDATE agent_run_log_stream SET {smuggled}, revision = revision + 1, last_modified_at = now() WHERE team_id = {{0}} AND id = {{1}}"; + + var raised = await Should.ThrowAsync(async () => + await scope.Resolve().Database.ExecuteSqlRawAsync(sql, [world.TeamId, stream])); + + PostgresErrorOf(raised).MessageText.ShouldContain(refusal); + var row = await StreamAsync(stream); + row.RetainUntil.ShouldBeNull("a refused statement leaves the row exactly as it was"); + row.PurgedAt.ShouldBeNull(); + } + + /// A live capture is never a retention candidate, and the guard is what makes that true rather than a query predicate somebody can forget. + [Fact] + public async Task The_guard_refuses_a_retention_statement_on_an_open_stream() + { + var world = await SeedWorldAsync(); + using var scope = _fixture.BeginScope(); + var logs = Logs(scope); + var session = Guid.NewGuid(); + var opened = (await logs.OpenAsync(Open(world, session), CancellationToken.None)).ShouldBeOfType(); + + var raised = await Should.ThrowAsync(async () => await scope.Resolve().Database.ExecuteSqlRawAsync( + "UPDATE agent_run_log_stream SET retain_until = now(), revision = revision + 1, last_modified_at = now() WHERE team_id = {0} AND id = {1}", + [world.TeamId, opened.Metadata.StreamId])); + + PostgresErrorOf(raised).MessageText.ShouldContain("rejected on a live stream"); + } + + /// A purge is final: nothing may put a stream back into a state where it claims to hold bytes it gave up. + [Fact] + public async Task The_guard_refuses_to_un_purge_a_purged_stream() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "a stream that will be reclaimed"); + await PurgeAsync(world, stream); + + using var scope = _fixture.BeginScope(); + var raised = await Should.ThrowAsync(async () => await scope.Resolve().Database.ExecuteSqlRawAsync( + "UPDATE agent_run_log_stream SET purged_at = NULL, revision = revision + 1, last_modified_at = now() WHERE team_id = {0} AND id = {1}", + [world.TeamId, stream])); + + PostgresErrorOf(raised).MessageText.ShouldContain("purge is final"); + } + + /// + /// Why this plane reclaims bytes and not rows, stated as the refusals themselves rather than as prose in a pull + /// request: all three archive tables reject deletion outright, so a tombstone on the surviving head is the only + /// legible way to say the bytes are gone. + /// + [Fact] + public async Task The_log_archive_tables_refuse_every_deletion() + { + var world = await SeedWorldAsync(); + var stream = await CaptureAsync(world, "an archive nothing may delete"); + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + + foreach (var (table, predicate, refusal) in new[] + { + ("agent_run_log_segment", "stream_id", "append-only"), + ("agent_run_log_verification", "stream_id", "durable history"), + ("agent_run_log_stream", "id", "DELETE rejected"), + }) + { + var raised = await Should.ThrowAsync(async () => + await db.Database.ExecuteSqlRawAsync($"DELETE FROM {table} WHERE {predicate} = {{0}}", [stream])); + + PostgresErrorOf(raised).MessageText.ShouldContain(refusal, customMessage: $"{table} is supposed to refuse deletion; if it no longer does, this plane could reclaim rows instead of tombstoning them"); + } + } + + // ── The world, and the durable streams inside it ─────────────────────────────────────────────────────────────── + + /// One real capture through the real service: an open, segments at a real local-rwx destination, a finalized source and a v3 completion. + private async Task CaptureAsync(World world, string text, string kind = AgentRunLogKinds.StandardOutput, int segments = 1, bool distinctSegments = true) + { + using var scope = _fixture.BeginScope(); + var logs = Logs(scope); + var session = Guid.NewGuid(); + var opened = (await logs.OpenAsync(Open(world, session, kind), CancellationToken.None)).ShouldBeOfType(); + var metadata = opened.Metadata; + + for (var ordinal = 1; ordinal <= segments; ordinal++) + { + // Distinct bytes per segment by default: the CAS is content-addressed, so repeating a payload makes two + // segments ONE object with a placement each — which is its own case, and the caller says when it wants it. + var suffix = distinctSegments ? $" #{ordinal:D4}" : string.Empty; + var bytes = System.Text.Encoding.UTF8.GetBytes($"{text}{suffix} {new string('x', 24)}"); + metadata = (await logs.AppendAsync(new AgentRunLogAppendRequest + { + TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = metadata.StreamId, WorkerFenceEpoch = Fence, + CaptureSessionId = session, ExpectedSegmentOrdinal = ordinal, ExpectedOffsetBytes = metadata.TotalBytes, + ExpectedSourceOffsetBytes = metadata.SourceOffsetBytes, SourceLengthBytes = bytes.Length, + StorageProfileId = world.StorageProfileId, StorageProfileRevision = 1, ActorId = world.ActorId, Bytes = bytes, + }, CancellationToken.None)).ShouldBeOfType().Metadata; + } + + var finalized = (await logs.FinalizeSourceAsync(new AgentRunLogFinalizeSourceRequest + { + TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = metadata.StreamId, WorkerFenceEpoch = Fence, + CaptureSessionId = session, ExpectedRevision = metadata.Revision, ExpectedSourceOffsetBytes = metadata.SourceOffsetBytes, + }, CancellationToken.None)).ShouldBeOfType(); + // Completion verifies a bounded number of segments per call, so a many-segment capture reports Progress until + // its manifest is whole — exactly what the production bridge does, and what a single call would silently miss. + var revision = finalized.Metadata.Revision; + + for (var pass = 0; pass < segments + 2; pass++) + { + var completion = await logs.CompleteAsync(new AgentRunLogCompleteRequest + { + TeamId = world.TeamId, AgentRunId = world.AgentRunId, StreamId = metadata.StreamId, WorkerFenceEpoch = Fence, + CaptureSessionId = session, ExpectedRevision = revision, + }, CancellationToken.None); + + if (completion is AgentRunLogCompleteResult.Completed) break; + + revision = completion.ShouldBeOfType().Metadata.Revision; + } + + (await StreamAsync(metadata.StreamId)).State.ShouldBe(AgentRunLogStreamState.Completed, "the capture this test starts from has to settle, or every assertion after it is about the wrong row"); + _staged.Add(new StagedStream(world.TeamId, metadata.StreamId)); + + return metadata.StreamId; + } + + /// + /// Time travel for the two clocks the claim query reads: how long ago the capture settled, and how long ago + /// anything last touched the row. Every timestamp on the row moves by the SAME delta, so their ordering — which + /// ck_agent_run_log_stream_time enforces and which is a real invariant rather than test scaffolding — still + /// holds; the shift is at least the recheck interval so that "age it by nothing" still reopens the deferral gate. + /// + /// The trigger is suspended for this one statement because the guard admits no rewrite of + /// completed_at at all — which is exactly the property the guard theory pins — and both windows are + /// measured in days no test can wait out. + /// + private async Task AgeAsync(Guid streamId, TimeSpan age) + { + var shift = age > TimeSpan.FromDays(2) ? age : TimeSpan.FromDays(2); + const string sql = "UPDATE agent_run_log_stream SET created_at = created_at - {0}::interval, capture_finalized_at = capture_finalized_at - {0}::interval, " + + "completed_at = completed_at - {0}::interval, last_modified_at = last_modified_at - {0}::interval WHERE id = {1}"; + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream DISABLE TRIGGER agent_run_log_stream_enforce_invariants"); + try { await db.Database.ExecuteSqlRawAsync(sql, [$"{shift.TotalSeconds} seconds", streamId]); } + finally { await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream ENABLE TRIGGER agent_run_log_stream_enforce_invariants"); } + } + + /// + /// Winds the clock past the quarantine the sweep itself wrote: the deadline lands in the past, and the row looks + /// settled two days ago rather than a moment ago, so the deferral gate is open again. Both are time travel — the + /// guard rightly refuses to move a modification time backwards — so it runs with the trigger suspended, exactly + /// like . That the reaper's OWN quarantine write is admitted by the guard is not assumed + /// here: it is what put a value in retain_until for this helper to move. + /// + private async Task ElapseQuarantineAsync(Guid streamId) + { + // GREATEST so the modification time never falls below the clocks ck_agent_run_log_stream_time orders it + // against: a stream whose capture was aged into the past is already older than the recheck interval anyway. + const string sql = "UPDATE agent_run_log_stream SET retain_until = now() - interval '1 second', revision = revision + 1, " + + "last_modified_at = GREATEST(created_at, capture_finalized_at, completed_at, last_modified_at - interval '2 days') WHERE id = {0}"; + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream DISABLE TRIGGER agent_run_log_stream_enforce_invariants"); + + try { (await db.Database.ExecuteSqlRawAsync(sql, [streamId])).ShouldBe(1); } + finally { await db.Database.ExecuteSqlRawAsync("ALTER TABLE agent_run_log_stream ENABLE TRIGGER agent_run_log_stream_enforce_invariants"); } + } + + /// Drives one stream all the way through the plane — quarantine, elapse, collect — for the tests whose subject is what happens AFTER a purge. + private async Task PurgeAsync(World world, Guid streamId) + { + await AgeAsync(streamId, TimeSpan.FromDays(31)); + var quarantine = await SweepAsync(); + await ElapseQuarantineAsync(streamId); + var collection = await SweepAsync(); + + (await StreamAsync(streamId)).PurgedAt.ShouldNotBeNull( + $"the premise of this test is a stream the plane actually reclaimed; quarantine pass {Describe(quarantine)}, collection pass {Describe(collection)}"); + (await UnpurgedLocationsAsync(world, streamId)).ShouldBe(0); + } + + /// + /// Takes every location behind this stream's objects out of the purge lifecycle without purging it. Deleted + /// is the state that says so: 0127's trigger makes it terminal, so unlike Missing — which a purge can still + /// drive to Purged — these bytes left by a path this plane will never drive. Its guards are suspended for + /// the one statement, exactly as the stream's are for time travel. + /// + private async Task MarkLocationsDeletedAsync(World world, Guid streamId) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + var objects = await ObjectsOfAsync(world, streamId); + await db.Database.ExecuteSqlRawAsync("ALTER TABLE artifact_location DISABLE TRIGGER USER"); + + try + { + await db.ArtifactLocation.Where(location => location.TeamId == world.TeamId && objects.Contains(location.ArtifactObjectId)) + .ExecuteUpdateAsync(set => set + .SetProperty(location => location.State, ArtifactLocationState.Deleted) + .SetProperty(location => location.Revision, location => location.Revision + 1)); + } + finally + { + await db.Database.ExecuteSqlRawAsync("ALTER TABLE artifact_location ENABLE TRIGGER USER"); + } + } + + /// An intent that has not finished and tries to name the object it is moving — the row the schema refuses. + private static async Task InFlightIntentAsync(CodeSpaceDbContext db, World world, Guid objectId) + { + var revisionId = await db.StorageProfileRevision.AsNoTracking() + .Where(row => row.StorageProfileId == world.StorageProfileId && row.Revision == 1).Select(row => row.Id).SingleAsync(); + + return new ArtifactTransferIntent + { + Id = Guid.NewGuid(), TeamId = world.TeamId, ArtifactObjectId = objectId, StorageProfileRevisionId = revisionId, + IdempotencyKey = $"retention-test/{objectId:N}", ExpectedDigest = System.Security.Cryptography.SHA256.HashData([7]), + ExpectedSizeBytes = 1, TargetLocator = "local-rwx", TargetObjectKey = $"transfer/{objectId:N}", + State = ArtifactTransferState.Intended, Revision = 1, CreatedDate = DateTimeOffset.UtcNow, CreatedBy = world.ActorId, + LastModifiedDate = DateTimeOffset.UtcNow, LastModifiedBy = world.ActorId, + }; + } + + // ── What a sealed result cites ───────────────────────────────────────────────────────────────────────────────── + + /// A cleanup receipt and an offloaded event payload for the same run, so the seal has all four pin kinds to write. + private async Task<(Guid ReceiptId, Guid ArtifactId)> SeedCitedRecordsAsync(World world) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + var artifactId = Guid.NewGuid(); + var receiptId = Guid.NewGuid(); + db.WorkflowArtifact.Add(new WorkflowArtifact + { + Id = artifactId, TeamId = world.TeamId, ContentType = "application/json", SizeBytes = 2, + Sha256 = Convert.ToHexString(System.Security.Cryptography.SHA256.HashData([1, 2])).ToLowerInvariant(), InlineBytes = [1, 2], + }); + db.AgentRunCleanupReceipt.Add(new AgentRunCleanupReceiptRecord + { + Id = receiptId, TeamId = world.TeamId, AgentRunId = world.AgentRunId, FenceEpoch = Fence, + Kind = Messages.Agents.Recovery.RunResourceKind.Spool, Outcome = Messages.Agents.Recovery.RunResourceOutcome.Completed, + RecordedByHost = "retention-test", RecordedAt = DateTimeOffset.UtcNow, + }); + await db.SaveChangesAsync(); + db.AgentRunEvent.Add(new AgentRunEvent + { + Id = Guid.NewGuid(), AgentRunId = world.AgentRunId, Kind = Messages.Agents.AgentEventKind.Warning, + Text = "an offloaded payload", DataArtifactId = artifactId, + }); + await db.SaveChangesAsync(); + + return (receiptId, artifactId); + } + + /// + /// A real seal through the real store: a hand-seeded protocol whose exact observation census is present, sealed as + /// a replay so the runtime gate — which is a different slice's concern — stays out of this one. + /// + private async Task SealResultAsync(World world) + { + var groupId = Guid.NewGuid(); + var manifest = new EvalSuiteManifest { Version = "sha256/retention-suite:v1", Cells = [new CorpusCellRef { TaskId = "task-a", Mode = BenchmarkMode.HarnessCli }] }; + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + db.PairedQualificationProtocol.Add(new PairedQualificationProtocol + { + ObservationGroupId = groupId, TeamId = world.TeamId, SuiteDigest = "sha256:hidden", SuiteVersion = manifest.Version, + CodeRevision = new string('c', 40), ControlModelRowId = world.ControlModelRowId, CandidateModelRowId = world.CandidateModelRowId, + RequiresCellAdmission = false, RequiresResultDigest = false, StatisticsVersion = PairedQualificationOutcome.StatisticsVersion, + Criterion = "Quality", SessionsPerCell = 1, MinimumIndependentClusters = 1, MinimumStrata = 1, MinimumRequiredExecutionClusters = 1, + MinimumEvaluatorHealth = 1, MaxCostUsdPerLaunch = 3m, MinimumQualityLift = 0.05, OrderingSeed = "frozen-order", + ProtocolDigest = Convert.ToHexString(System.Security.Cryptography.SHA256.HashData(groupId.ToByteArray())), + }); + foreach (var arm in new[] { "control", "candidate" }) + db.BenchmarkResultRecord.Add(Observation(world, groupId, manifest.Version, arm)); + await db.SaveChangesAsync(); + + var outcome = Outcome(groupId, manifest.Version, db.PairedQualificationProtocol.Single(row => row.ObservationGroupId == groupId).ProtocolDigest); + var store = new PairedQualificationResultStore(db, scope.Resolve()); + await store.SealAsync(new PairedQualificationSealRequest + { + ObservationGroupId = groupId, Manifest = manifest, Outcome = outcome, Source = PairedQualificationSealSource.Replay, + }, CancellationToken.None); + + return groupId; + } + + private BenchmarkResultRecord Observation(World world, Guid groupId, string suiteVersion, string arm) => new() + { + Id = Guid.NewGuid(), TeamId = world.TeamId, SuiteVersion = suiteVersion, TaskId = "task-a", Mode = BenchmarkMode.HarnessCli.ToString(), + Harness = "claude-code", Model = arm, ModelCredentialModelId = arm == "control" ? world.ControlModelRowId : world.CandidateModelRowId, + ObservationGroupId = groupId, ObservationArm = arm, ObservationSession = 0, AgentRunId = world.AgentRunId, + ObservedModel = arm, OutcomeState = "Solved", Solved = true, RunStatus = AgentRunStatus.Succeeded.ToString(), + MaxCostUsd = 3m, CostUsd = 1m, CostIndeterminate = false, GitSha = new string('c', 40), + }; + + private static PairedQualificationOutcome Outcome(Guid groupId, string suiteVersion, string protocolDigest) => new() + { + ObservationGroupId = groupId, ProtocolDigest = protocolDigest, CodeRevision = new string('c', 40), SuiteDigest = "sha256:hidden", + SuiteVersion = suiteVersion, IndependentClusters = 1, PairedCells = 1, RequiredExecutionClusters = 1, + Control = Arm(), Candidate = Arm(), QualityDifference = 0.1, QualityDifferenceLower95 = 0.06, + RequiredExecutionComplete = true, QualifiedForCapabilityClaim = true, BlockingReasons = [], Strata = [], InfraFailures = [], + }; + + private static PairedArmQualificationSummary Arm() => new() + { + Solved = 1, BudgetAdmissibleSolved = 1, Total = 1, InfraUnknown = 0, CostKnownCells = 1, + CapabilityVerdictCells = 1, ObservedModelCells = 1, EvaluatorHealth = 1, ObservedModels = ["model"], + }; + + private static string Identity(PairedQualificationResultPin pin) => $"{pin.Kind}:{pin.Target}"; + + private async Task ForgetPinsAsync(Guid resultId) + { + using var scope = _fixture.BeginScope(); + await scope.Resolve().PairedQualificationResultPin.Where(pin => pin.ResultId == resultId).ExecuteDeleteAsync(); + } + + /// + /// Runs migration 0235's own backfill statements, read from the script that ships with the build rather than + /// copied here — a mirror would keep passing after the migration changed underneath it. + /// + private async Task RunBackfillAsync() + { + var script = await File.ReadAllTextAsync(Path.Combine(AppContext.BaseDirectory, "Persistence", "DbUpFiles", "0235_durable_retention_pins.sql")); + var statements = script.Split(';').Select(value => value.Trim()).Where(value => value.Contains("INSERT INTO paired_qualification_result_pin", StringComparison.Ordinal)).ToList(); + + statements.Count.ShouldBe(4, "0235 backfills four closures — runs, then their streams, receipts and offloaded payloads; if that count changed, so did what a pre-existing result protects"); + using var scope = _fixture.BeginScope(); + + foreach (var statement in statements) + await scope.Resolve().Database.ExecuteSqlRawAsync(statement[statement.IndexOf("INSERT INTO", StringComparison.Ordinal)..]); + } + + // ── Reads ────────────────────────────────────────────────────────────────────────────────────────────────────── + + private static string Describe(DurableRetentionSweepSummary summary) => + $"[claimed {summary.Claimed}, quarantined {summary.Quarantined}, collected {summary.Collected}, referenced {summary.Referenced}, kept {summary.Kept}]"; + + private async Task SweepAsync() + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().SweepAsync(CancellationToken.None); + } + + /// + /// The real reaper over the real cursor, with only the CAS purge verdict swapped for the one a destination that + /// will not give the bytes up produces. Everything that decides whether the tombstone is written is production + /// code; the refusal is the single injected fact. + /// + private async Task SweepAgainstARefusingDestinationAsync() + { + using var scope = _fixture.BeginScope(); + var options = scope.Resolve>(); + var cursor = new LogStreamRetentionCursor(options, new RefusingPurgeCoordinator(), NullLogger.Instance); + + return await new DurableRetentionReaper(options, [cursor], NullLogger.Instance).SweepAsync(CancellationToken.None); + } + + /// A destination that answers every delete with a refusal and no effect — the shape a revoked key or a read-only bucket produces. + private sealed class RefusingPurgeCoordinator : IArtifactCasPurgeCoordinator + { + public Task PurgeAsync(ArtifactCasPurgeRequest request, CancellationToken cancellationToken) => + Task.FromResult(new ArtifactCasPurgeResult.Rejected(new ArtifactCasProblem(ArtifactCasProblemCode.ProviderUnavailable, true))); + + public Task ClaimAsync(ArtifactCasPurgeRequest request, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task DeleteAsync(ArtifactCasPurgeClaim claim, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task ReleaseAsync(ArtifactCasPurgeClaim claim, ArtifactCasReleaseEvidence evidence, CancellationToken cancellationToken) => throw new NotSupportedException(); + public Task AbandonAsync(ArtifactCasPurgeClaim claim, CancellationToken cancellationToken) => throw new NotSupportedException(); + } + + private LogStreamRetentionCursor Cursor() + { + using var scope = _fixture.BeginScope(); + + return new LogStreamRetentionCursor(scope.Resolve>(), scope.Resolve(), NullLogger.Instance); + } + + /// What the production claim query returns for the window the loop would build right now. + private async Task> CandidatesAsync(int limit) + { + var rule = DurableRetentionPolicy.LogStream; + var now = DateTimeOffset.UtcNow; + + return await Cursor().ClaimAsync(new DurableRetentionSweepWindow(now, now - rule.MinimumAge, now - rule.RecheckInterval), limit, CancellationToken.None); + } + + private async Task> ClaimAsync(int limit) => (await CandidatesAsync(limit)).Select(candidate => candidate.Id).ToList(); + + private async Task StreamAsync(Guid streamId) + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().AgentRunLogStream.AsNoTracking().SingleAsync(row => row.Id == streamId); + } + + private async Task> PinsAsync(Guid resultId) + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().PairedQualificationResultPin.AsNoTracking() + .Where(pin => pin.ResultId == resultId).OrderBy(pin => pin.Kind).ToListAsync(); + } + + /// The ids of the transactions that inserted the result and its pins. One id is the whole atomicity claim. + private async Task> InsertingTransactionsAsync(Guid resultId) + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().Database.SqlQueryRaw( + "SELECT DISTINCT xmin::text AS \"Value\" FROM paired_qualification_result WHERE observation_group_id = {0} " + + "UNION SELECT DISTINCT xmin::text FROM paired_qualification_result_pin WHERE result_id = {0}", resultId).ToListAsync(); + } + + private async Task> ObjectsOfAsync(World world, Guid streamId) + { + using var scope = _fixture.BeginScope(); + + return await scope.Resolve().AgentRunLogSegment.AsNoTracking() + .Where(segment => segment.TeamId == world.TeamId && segment.StreamId == streamId) + .Select(segment => segment.ArtifactObjectId).Distinct().OrderBy(id => id).ToListAsync(); + } + + private async Task UnpurgedLocationsAsync(World world, Guid streamId) + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + + return await db.ArtifactLocation.AsNoTracking() + .Where(location => location.TeamId == world.TeamId && location.State != ArtifactLocationState.Purged && location.State != ArtifactLocationState.Deleted) + .Join(db.AgentRunLogSegment.AsNoTracking().Where(segment => segment.StreamId == streamId), + location => location.ArtifactObjectId, segment => segment.ArtifactObjectId, (location, _) => location.Id) + .Distinct() + .CountAsync(); + } + + private async Task MetadataAsync(World world, Guid streamId) + { + using var scope = _fixture.BeginScope(); + + return (await Logs(scope).GetMetadataAsync(world.TeamId, streamId, CancellationToken.None)).ShouldBeOfType().Metadata; + } + + private async Task ReadAsync(World world, Guid streamId) + { + using var scope = _fixture.BeginScope(); + var stream = await StreamAsync(streamId); + + return await Logs(scope).ReadRangeAsync(new AgentRunLogRangeRequest(world.TeamId, streamId, 0, checked((int)stream.TotalBytes)), CancellationToken.None); + } + + /// Whether the stream's bytes still come back through the production read path — the only reading of "the bytes are there" that matters. + private async Task BytesReadableAsync(World world, Guid streamId) => await ReadAsync(world, streamId) is AgentRunLogRangeResult.Available; + + private static RoomProjector.AgentLogRow Row(AgentRunLogStream stream) => + new(stream.AgentRunId, stream.State, stream.SchemaVersion, stream.ManifestDigest != null, stream.RemoteStallSince != null, stream.PurgedAt != null); + + private static AgentRunLogService Logs(ILifetimeScope scope) => + new(scope.Resolve>(), scope.Resolve(), TimeProvider.System); + + private static AgentRunLogOpenRequest Open(World world, Guid session, string kind = AgentRunLogKinds.StandardOutput) => new() + { + TeamId = world.TeamId, AgentRunId = world.AgentRunId, WorkerFenceEpoch = Fence, CaptureSessionId = session, + StreamKind = kind, ContentType = AgentRunLogRepresentations.PlainTextContentType, + ContentEncoding = AgentRunLogRepresentations.Utf8ContentEncoding, CaptureSource = "retention-test/v1", + }; + + /// EF's execution strategy wraps the server error; the guard's own message stays decisive. + private static PostgresException PostgresErrorOf(Exception error) + { + while (error is not PostgresException && error.InnerException is { } inner) error = inner; + return error.ShouldBeOfType(); + } + + private const long Fence = 7; + + /// A team whose log storage route is Active, whose credential resolves, and whose local-rwx root is real — so a refused purge is only ever the one the test caused. + private async Task SeedWorldAsync() + { + var actorId = Guid.NewGuid(); + var teamId = Guid.NewGuid(); + var profileId = Guid.NewGuid(); + var credentialId = Guid.NewGuid(); + var now = DateTimeOffset.UtcNow; + var root = NewRoot(); + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + db.User.Add(new User { Id = actorId, Email = $"durable-retention-{actorId:N}@test.local", Name = "Durable Retention" }); + db.Team.Add(new Team { Id = teamId, Slug = $"durable-retention-{teamId:N}", Name = "Durable Retention", Kind = TeamKind.Workspace }); + db.TeamMembership.Add(new TeamMembership { Id = Guid.NewGuid(), TeamId = teamId, UserId = actorId, Role = TeamRole.Owner }); + var credential = new StorageCredential + { + Id = credentialId, TeamId = teamId, StableName = $"durable-retention-{credentialId:N}", + CurrentRevision = 1, State = StorageCredentialState.Active, CreatedDate = now, CreatedBy = actorId, + }; + credential.Revisions.Add(new StorageCredentialRevision + { + Id = Guid.NewGuid(), TeamId = teamId, StorageCredentialId = credentialId, Revision = 1, + ProviderTypeKey = LocalRwxArtifactStorageDriverFactory.TypeKey, EncryptedPayload = scope.Resolve().Encrypt("{}"), + SafeHint = "safe", EnvelopeFingerprint = $"sha256:{new string('b', 64)}", CreatedDate = now, CreatedBy = actorId, + }); + db.StorageCredential.Add(credential); + var profile = new StorageProfile + { + Id = profileId, TeamId = teamId, StableName = $"durable-retention-{profileId:N}", State = StorageProfileState.Active, + CurrentRevision = 1, CreatedDate = now, CreatedBy = actorId, LastModifiedDate = now, LastModifiedBy = actorId, + }; + profile.Revisions.Add(new StorageProfileRevision + { + Id = Guid.NewGuid(), TeamId = teamId, StorageProfileId = profileId, Revision = 1, + ProviderTypeKey = LocalRwxArtifactStorageDriverFactory.TypeKey, NonSecretConfigJson = $"{{\"rootPath\":\"{root.Replace("\\", "\\\\")}\"}}", + CredentialRef = $"db:{credentialId:D}:1", NamespaceFingerprint = $"sha256:{Convert.ToHexString(System.Security.Cryptography.SHA256.HashData(profileId.ToByteArray())).ToLowerInvariant()}", + CreatedDate = now, CreatedBy = actorId, + }); + db.StorageProfile.Add(profile); + await db.SaveChangesAsync(); + var runId = Guid.NewGuid(); + db.AgentRun.Add(new AgentRun + { + Id = runId, TeamId = teamId, Harness = "test-harness", Status = AgentRunStatus.Running, TaskJson = "{}", + FenceEpoch = Fence, CreatedDate = now, CreatedBy = actorId, LastModifiedDate = now, LastModifiedBy = actorId, + }); + await db.SaveChangesAsync(); + + return new World(teamId, actorId, profileId, runId, Guid.NewGuid(), Guid.NewGuid()); + } + + private string NewRoot() + { + var root = Path.Combine(Path.GetTempPath(), $"codespace-durable-retention-{Guid.NewGuid():N}"); + Directory.CreateDirectory(root); + _roots.Add(root); + return root; + } + + public Task InitializeAsync() => Task.CompletedTask; + + /// + /// Reclaims every placement this class staged, through the same purge lifecycle the production plane uses, BEFORE + /// the destinations it wrote them to are removed. + /// + /// This is not tidiness. The artifact-location verifier is a bounded deployment-wide sweep — the hundred + /// least-recently-verified placements across every team, on a database this collection shares — and its own tests + /// assert that it reached THEIR rows. Live placements left behind a deleted local root change both what that batch + /// contains and which destinations answer in it, so a class that stages placements owns them all the way to the + /// end. + /// + public async Task DisposeAsync() + { + foreach (var staged in _staged) await ReclaimAsync(staged); + + foreach (var root in _roots) + { + try { if (Directory.Exists(root)) Directory.Delete(root, recursive: true); } catch { /* best-effort */ } + } + } + + private async Task ReclaimAsync(StagedStream staged) + { + try + { + using var scope = _fixture.BeginScope(); + var db = scope.Resolve(); + var purge = scope.Resolve(); + var objects = db.AgentRunLogSegment.AsNoTracking() + .Where(segment => segment.TeamId == staged.TeamId && segment.StreamId == staged.StreamId) + .Select(segment => segment.ArtifactObjectId).Distinct(); + + // Per LOCATION, not per object: identical captures deduplicate to one object with a placement per write, + // and a claim that does not name which placement refuses a replicated object outright — which is the + // production behaviour, and the reason this cleanup has to be explicit about it. + var placements = await db.ArtifactLocation.AsNoTracking() + .Where(location => location.TeamId == staged.TeamId && objects.Contains(location.ArtifactObjectId) + && location.State != ArtifactLocationState.Purged && location.State != ArtifactLocationState.Deleted) + .Select(location => new { location.Id, location.ArtifactObjectId }) + .ToListAsync(); + + foreach (var placement in placements) + await purge.PurgeAsync(new ArtifactCasPurgeRequest + { + TeamId = staged.TeamId, ArtifactObjectId = placement.ArtifactObjectId, + ArtifactLocationId = placement.Id, ActorId = Messages.Constants.SystemUsers.SeederId, + }, CancellationToken.None); + } + catch { /* best-effort: a test that already purged, or left a refusing destination, has nothing more to give back */ } + } + + private sealed record StagedStream(Guid TeamId, Guid StreamId); + + private sealed record World(Guid TeamId, Guid ActorId, Guid StorageProfileId, Guid AgentRunId, Guid ControlModelRowId, Guid CandidateModelRowId); +} diff --git a/backend/tests/CodeSpace.UnitTests/Architecture/RequestAuthorizationInventoryTests.cs b/backend/tests/CodeSpace.UnitTests/Architecture/RequestAuthorizationInventoryTests.cs index 6ab9a89d4..eda533c36 100644 --- a/backend/tests/CodeSpace.UnitTests/Architecture/RequestAuthorizationInventoryTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Architecture/RequestAuthorizationInventoryTests.cs @@ -64,6 +64,7 @@ public class RequestAuthorizationInventoryTests ["ReapAgentRunSpoolsCommand"] = "sweep", ["ReapAgentRunOrphansCommand"] = "settles the cleanup receipts addressed to THIS host, across every team, by tearing down or observing host-local resources agent runs left behind when they were abandoned from another worker; which kernel objects and directories a machine still holds is a question no team can ask on its own behalf, and it reclaims what a run already finished with rather than changing any run's outcome.", ["ReapUnreferencedArtifactsCommand"] = "collects artifacts that a producer declared for retention and that no reference site points at, in bounded lease/fence batches; it is dispatched only by the system recurring job, acts for no user, and can never reach an artifact no producer declared.", + ["ReapExpiredDurableRecordsCommand"] = "reclaims the bytes behind durable records whose retention rule elapsed and that nothing — no column, no qualification pin — still cites, across every team in bounded batches; how long an archive is kept is a deployment-wide policy no team states on its own behalf, and it removes only what both waits and a fail-closed citation probe already cleared.", ["ReconcileAgentRunLogCapturesCommand"] = "reconciles exact durable AgentRun log-capture health in bounded lease/fence batches; it is dispatched only by the system recurring job and never acts for a user or changes an AgentRun outcome.", ["ResumeAbandonedArtifactTransfersCommand"] = "finishes artifact transfers whose worker died mid-flight, across every team, in bounded fence/lease batches; a parked transfer is unreachable to the team that started it, so this is a question no team can ask on its own behalf, and it starts nothing — it only completes or closes what a caller already began.", ["ReconcileRunDataManifestsCommand"] = "un-states the expectations of terminal runs whose producers declared more records than any of them accounted for, across every team, in bounded batches; whether a run's record was ever established is a question no team can ask on its own behalf, and it removes a claim rather than making one — no run can come out of it reading more complete than it went in.", diff --git a/backend/tests/CodeSpace.UnitTests/Persistence/AgentRunLogSchemaTests.cs b/backend/tests/CodeSpace.UnitTests/Persistence/AgentRunLogSchemaTests.cs index 48457f7be..e72c4f306 100644 --- a/backend/tests/CodeSpace.UnitTests/Persistence/AgentRunLogSchemaTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Persistence/AgentRunLogSchemaTests.cs @@ -28,6 +28,11 @@ public void Stream_is_a_tenant_bound_monotonic_head_for_one_open_versioned_log_k "AgentRunId", "CaptureFinalizedAt", "CaptureSessionId", "CaptureSource", "CaptureSourceBaseOffsetBytes", "CompletedAt", "ContentDigest", "ContentDigestAlgorithm", "ContentEncoding", "ContentType", "CreatedAt", "ErrorCode", "ErrorMessage", "ExpiresAt", "Id", "LastModifiedAt", "ManifestDigest", "NextOffsetBytes", "NextSegmentOrdinal", "RemoteStallCode", "RemoteStallSince", "Retention", "Revision", "SchemaVersion", "SegmentCount", "State", + // The retention plane's two columns (migration 0235), and the only part of this row a process that is not + // the capturing worker may ever write: retain_until is the instant the bytes become reclaimable, purged_at + // the tombstone the head row keeps after they are gone. The guard admits them on a TERMINAL stream only, + // alone, and never in the same statement as anything else — so a purge can never pass for a capture verdict. + "PurgedAt", "RetainUntil", "SourceOffsetBytes", "StreamKind", "TeamId", "TotalBytes", "WorkerFenceEpoch", "Xmin", }.Order()); entity.FindProperty(nameof(AgentRunLogStream.State))!.GetMaxLength().ShouldBe(24); diff --git a/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomLogSummaryTests.cs b/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomLogSummaryTests.cs index 49eda2238..ad75b02bb 100644 --- a/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomLogSummaryTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Sessions/Room/RoomLogSummaryTests.cs @@ -123,6 +123,32 @@ public void A_terminal_stream_still_outranks_a_held_one() summary.Detail.ShouldBe("2 streams · 1 capture failed · 1 held; storage unavailable"); } - private static RoomProjector.AgentLogRow Row(AgentRunLogStreamState state, int schemaVersion = 3, bool hasManifestDigest = false, bool remoteStalled = false) => - new(AgentId, state, schemaVersion, hasManifestDigest, remoteStalled); + /// A stream the retention plane reclaimed reads as purged, never as the integrity proof its manifest receipt still carries. + [Fact] + public void A_purged_stream_is_not_folded_as_integrity_verified() + { + var summary = RoomProjector.SummarizeLogs([Row(AgentRunLogStreamState.Completed, schemaVersion: 3, hasManifestDigest: true, purged: true)]); + + summary.ShouldBe(new RoomAgentLogSummary(RoomAgentLogStatus.Purged, 1, "1 stream · 1 purged; retention window elapsed")); + } + + [Fact] + public void A_readable_stream_outranks_a_purged_one_but_a_finalizing_one_outranks_both() + { + var purgedAndVerified = RoomProjector.SummarizeLogs([ + Row(AgentRunLogStreamState.Completed, schemaVersion: 3, hasManifestDigest: true, purged: true), + Row(AgentRunLogStreamState.Completed, schemaVersion: 3, hasManifestDigest: true), + ]); + + purgedAndVerified.Status.ShouldBe(RoomAgentLogStatus.Purged, "a reader has to be told some of this agent's log is gone, even when the rest verifies"); + purgedAndVerified.Detail.ShouldBe("2 streams · 1 purged; retention window elapsed · 1 integrity verified"); + + RoomProjector.SummarizeLogs([ + Row(AgentRunLogStreamState.Completed, schemaVersion: 3, hasManifestDigest: true, purged: true), + Row(AgentRunLogStreamState.Open), + ]).Status.ShouldBe(RoomAgentLogStatus.Finalizing, "a capture still in flight is the more urgent fact"); + } + + private static RoomProjector.AgentLogRow Row(AgentRunLogStreamState state, int schemaVersion = 3, bool hasManifestDigest = false, bool remoteStalled = false, bool purged = false) => + new(AgentId, state, schemaVersion, hasManifestDigest, remoteStalled, purged); } diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs index 13d718d15..3a397dfbd 100644 --- a/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Artifacts/ArtifactReferenceOracleTests.cs @@ -38,6 +38,7 @@ public void Every_soft_link_to_an_artifact_is_a_probed_site() ("workflow_run_tool_call_attempt", "result_artifact_id"), ("workflow_run_tool_call_attempt", "error_artifact_id"), ("workflow_run_sensitive_record_payload", "ciphertext_artifact_id"), + ("paired_qualification_result_pin", "pinned_artifact_id"), }, ignoreOrder: true); } diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs new file mode 100644 index 000000000..6a93f6cf8 --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/DurableRetentionPolicyTests.cs @@ -0,0 +1,155 @@ +using CodeSpace.Core.Services.Workflows.Retention; +using CodeSpace.Messages.Constants; +using CodeSpace.Messages.Retention; +using Shouldly; + +namespace CodeSpace.UnitTests.Workflows.Retention; + +/// +/// The committed rule table and the one function that decides whether a durable record may be reclaimed. The windows +/// are asserted as literals on purpose: they are the only numbers in this plane whose reduction destroys data, so +/// shortening one has to be a deliberate edit here as well as in the policy. +/// +/// Every test below names the mutation it catches, because a retention decision that is wrong in the permissive +/// direction has no second chance — the bytes are gone. +/// +[Trait("Category", "Unit")] +public sealed class DurableRetentionPolicyTests +{ + private static readonly DateTimeOffset Now = new(2026, 9, 18, 12, 0, 0, TimeSpan.Zero); + private static readonly DurableRetentionRule Rule = DurableRetentionPolicy.LogStream; + + [Fact] + public void The_committed_rule_table_is_pinned_to_its_literal_windows() + { + var rule = DurableRetentionPolicy.For(DurableRecordClass.LogStream).ShouldNotBeNull(); + + rule.MinimumAge.ShouldBe(TimeSpan.FromDays(30), "a log stream is not a candidate until a month after its capture settled"); + rule.QuarantineWindow.ShouldBe(TimeSpan.FromHours(24), "a second, independent day passes after the first uncited observation"); + rule.RecheckInterval.ShouldBe(TimeSpan.FromHours(24), "a stream that was looked at and kept is left alone this long, so one unreclaimable row cannot own a batch slot"); + } + + [Fact] + public void Every_declared_class_has_a_rule_and_the_table_declares_nothing_else() + { + // The table advertises what is actually reclaimed. A class listed here without a cursor would read as a + // promise the system does not keep; a cursor whose class is missing claims nothing at all. Either way the + // right time to notice is here. + DurableRetentionPolicy.Rules.Keys.Order().ShouldBe(Enum.GetValues().Order(), + customMessage: "a class with no rule is never claimed, so adding one to the enum without a rule silently disables its plane"); + Enum.GetValues().ShouldBe([DurableRecordClass.LogStream], + customMessage: "a class belongs here only together with the cursor that sweeps it — add both in one change, never the rule first"); + } + + /// + /// Who a reclamation is attributed to. The seeder is the established identity for background work — Hangfire + /// workers, scheduled jobs and DbUp all write under it (SystemUsers.cs), and it holds a real seeded row, so + /// a purge receipt attributes to something that exists. A dedicated retention actor would be more specific, but it + /// would need its own seeded user and migration to be more HONEST; until then the choice is pinned here so + /// changing it is a decision rather than a diff. + /// + [Fact] + public void A_reclamation_is_attributed_to_the_system_background_actor() + { + SystemUsers.SeederId.ShouldBe(Guid.Parse("00000000-0000-0000-0000-000000000001")); + SystemUsers.SeederName.ShouldBe("System"); + } + + [Fact] + public void A_class_the_policy_does_not_register_has_no_rule_and_therefore_keeps() + { + // The reaper reads a null rule as "claim nothing"; the decision reads it as Indeterminate. Both mean keep, + // which is what makes REMOVING a class from the table a safe operation rather than a purge. + DurableRetentionPolicy.For((DurableRecordClass)9999).ShouldBeNull(); + Decide(null, Now.AddDays(-400), null, DurableReferenceVerdict.Unreferenced).Action.ShouldBe(DurableRetentionAction.Indeterminate); + } + + [Fact] + public void A_citation_clears_the_quarantine_it_contradicts() + { + // Mutation: carry the old retain_until through a Referenced settlement. A stream cited for a year would then + // be collectable the instant its pin went away, with a quarantine "window" that elapsed while something still + // pointed at it — a second wait that never actually waited for anything. + var decision = Decide(Rule, Now.AddDays(-400), Now.AddDays(-100), DurableReferenceVerdict.Referenced); + + decision.RetainUntil.ShouldBeNull(); + } + + [Fact] + public void An_unanswered_question_keeps_the_quarantine_it_never_contradicted() + { + // The opposite of the test above, and the reason the two are separate: an unreadable citation site says + // nothing about the earlier uncited observation, so clearing the marker would restart a wait that was already + // most of the way through. + var quarantinedAt = Now.AddHours(-1); + var decision = Decide(Rule, Now.AddDays(-400), quarantinedAt, DurableReferenceVerdict.Indeterminate); + + decision.RetainUntil.ShouldBe(quarantinedAt); + } + + [Fact] + public void Pinned_row_is_never_collected() + { + // Mutation: drop the pin check in the cursor (so a pinned record classifies Unreferenced) and this arm is the + // only thing left between a sealed qualification result and the evidence it was computed from. + var decision = Decide(Rule, Now.AddDays(-400), Now.AddDays(-100), DurableReferenceVerdict.Referenced); + + decision.Action.ShouldBe(DurableRetentionAction.Referenced, "a citation outranks every elapsed window, including a quarantine that is long past"); + decision.RetainUntil.ShouldBeNull(); + } + + [Fact] + public void First_unreferenced_observation_only_quarantines() + { + // Mutation: collect on the first uncited observation. The record would then be removed by the very sweep that + // first looked at it, with no durable window in which an operator or a late writer could intervene. + var decision = Decide(Rule, Now.AddDays(-400), null, DurableReferenceVerdict.Unreferenced); + + decision.Action.ShouldBe(DurableRetentionAction.Quarantine); + decision.RetainUntil.ShouldBe(Now.Add(Rule.QuarantineWindow)); + } + + [Fact] + public void An_open_quarantine_window_waits_rather_than_collecting() + { + var decision = Decide(Rule, Now.AddDays(-400), Now.AddHours(1), DurableReferenceVerdict.Unreferenced); + + decision.Action.ShouldBe(DurableRetentionAction.Wait); + decision.Code.ShouldBe("quarantine-window-open"); + } + + [Fact] + public void Both_waits_elapsed_with_no_citation_is_the_only_path_to_collect() + { + var decision = Decide(Rule, Now.AddDays(-400), Now.AddSeconds(-1), DurableReferenceVerdict.Unreferenced); + + decision.Action.ShouldBe(DurableRetentionAction.Collect); + } + + [Theory] + [InlineData(DurableReferenceVerdict.Unreferenced)] + [InlineData(DurableReferenceVerdict.Referenced)] + [InlineData(DurableReferenceVerdict.Indeterminate)] + public void Young_row_keeps_whatever_its_verdict(DurableReferenceVerdict verdict) + { + // Mutation: remove the age floor. A record whose citing write is still in flight — a seal mid-transaction, a + // receipt about to be upserted — would be observed uncited and quarantined before its writer ever committed. + var decision = Decide(Rule, Now.AddDays(-1), Now.AddDays(-100), verdict); + + decision.Action.ShouldBeOneOf(DurableRetentionAction.Wait, DurableRetentionAction.Referenced); + decision.Action.ShouldNotBe(DurableRetentionAction.Collect); + } + + [Fact] + public void An_unanswered_citation_question_keeps_the_record() + { + // Mutation: treat Indeterminate as Unreferenced. "I could not tell" must never resolve to "delete". + var decision = Decide(Rule, Now.AddDays(-400), Now.AddDays(-100), DurableReferenceVerdict.Indeterminate); + + decision.Action.ShouldBe(DurableRetentionAction.Indeterminate); + decision.Code.ShouldBe("reference-status-indeterminate"); + } + + private static DurableRetentionDecision Decide(DurableRetentionRule? rule, DateTimeOffset terminalAt, DateTimeOffset? retainUntil, DurableReferenceVerdict verdict) => + DurableRetentionDecision.Decide(rule, new DurableRetentionObservation(terminalAt, retainUntil, verdict, Now)); +} diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs new file mode 100644 index 000000000..d5d46dd93 --- /dev/null +++ b/backend/tests/CodeSpace.UnitTests/Workflows/Retention/LogStreamCitationSitesTests.cs @@ -0,0 +1,123 @@ +using CodeSpace.Core.Persistence.Db; +using CodeSpace.Core.Persistence.Entities; +using CodeSpace.Core.Services.Workflows.Retention.Cursors; +using Microsoft.EntityFrameworkCore; +using Shouldly; + +namespace CodeSpace.UnitTests.Workflows.Retention; + +/// +/// The enumeration of everything that can still name a log stream's CAS object. It is pinned for the same reason +/// ArtifactReferenceOracle.ReferenceSites is: no foreign key points at an artifact_object from most of +/// these places, so the list IS the completeness argument, and a citer missing from it makes the cursor answer +/// "unreferenced" about bytes something still reaches. +/// +/// Two checks keep the list from being decoration. One reads the EF model rather than this list, so a future +/// column that names an artifact object reds here even though nobody thought about retention while adding it. The +/// other reads the cursor's own source, so an entry ADDED to the list without a probe beside it reds too — a pinned +/// list nothing cross-checks is a comment with a test around it. +/// +[Trait("Category", "Unit")] +public sealed class LogStreamCitationSitesTests +{ + private const string UnreachableDatabase = "Host=127.0.0.1;Port=1;Database=unused;Username=unused;Password=unused"; + + [Fact] + public void Every_citer_of_a_log_streams_bytes_is_pinned() + { + LogStreamRetentionCursor.CitationSites.ShouldBe(new[] + { + // The qualification pin: a sealed result still cites this stream. + ("paired_qualification_result_pin", "pinned_id"), + + // Another stream's segment. The CAS is content-addressed, so two captures of identical output — the + // ordinary case for short or empty streams — are ONE object under two names. + ("agent_run_log_segment", "artifact_object_id"), + + // An artifact stored as the same routed object. + ("workflow_artifact", "cas_artifact_object_id"), + }, ignoreOrder: true); + } + + [Fact] + public void A_mapped_column_that_names_an_artifact_object_and_is_not_probed_fails_this_test() + { + using var db = BuildContext(); + var probed = LogStreamRetentionCursor.CitationSites.Select(site => $"{site.Table}.{site.Column}").ToHashSet(StringComparer.Ordinal); + + var mapped = db.Model.GetEntityTypes() + .SelectMany(entity => entity.GetProperties().Select(property => (Table: entity.GetTableName(), Column: property.GetColumnName()))) + .Where(column => column.Table is not null && column.Column is not null && IsObjectSoftLink(column.Table!, column.Column!)) + .Select(column => $"{column.Table}.{column.Column}") + .Distinct(StringComparer.Ordinal) + .Order(StringComparer.Ordinal) + .ToList(); + + mapped.Where(column => !probed.Contains(column)).ShouldBeEmpty( + "a column that names an artifact_object which the log-stream cursor does not probe would let it purge bytes that row still reaches — " + + $"add it to {nameof(LogStreamRetentionCursor)}.{nameof(LogStreamRetentionCursor.CitationSites)} and to the probe beside it"); + } + + /// + /// Every entry in the list has a probe beside it, checked against the cursor's own source: a site is only a site + /// if something asks it. The check is textual on purpose — the probes are LINQ over EF entities, so there is no + /// runtime handle to count, and the alternative (trusting the list) is what this test exists to refuse. + /// + [Fact] + public void Every_pinned_citation_site_is_actually_probed_by_the_cursor() + { + using var db = BuildContext(); + var source = File.ReadAllText(Path.Combine(ProductionSourceRoot(), "CodeSpace.Core", "Services", "Workflows", "Retention", "Cursors", $"{nameof(LogStreamRetentionCursor)}.cs")); + + var unprobed = LogStreamRetentionCursor.CitationSites + .Select(site => (site.Table, site.Column, Member: MemberOf(db, site.Table, site.Column))) + .Where(site => !source.Contains($".{site.Member}", StringComparison.Ordinal)) + .Select(site => $"{site.Table}.{site.Column} (no use of .{site.Member})") + .ToList(); + + unprobed.ShouldBeEmpty( + $"a site listed in {nameof(LogStreamRetentionCursor.CitationSites)} that nothing reads is a claim the cursor does not keep — " + + "add the probe, or take the entry out:\n " + string.Join("\n ", unprobed)); + } + + /// The CLR property a column is mapped from — the name the cursor's LINQ has to mention if it reads that column at all. + private static string MemberOf(CodeSpaceDbContext db, string table, string column) + { + var entity = db.Model.GetEntityTypes().FirstOrDefault(type => type.GetTableName() == table) + ?? throw new InvalidOperationException($"'{table}' is not mapped, so a citation site names a table this build does not have."); + + return (entity.GetProperties().FirstOrDefault(property => property.GetColumnName() == column) + ?? throw new InvalidOperationException($"'{table}.{column}' is not mapped, so a citation site names a column this build does not have.")).Name; + } + + private static string ProductionSourceRoot() + { + for (var directory = new DirectoryInfo(AppContext.BaseDirectory); directory is not null; directory = directory.Parent) + { + var candidate = Path.Combine(directory.FullName, "backend", "src"); + if (Directory.Exists(candidate)) return candidate; + } + + throw new DirectoryNotFoundException($"'backend/src' was not found above '{AppContext.BaseDirectory}', so the probes were never checked. Run the unit suite from the repository checkout."); + } + + /// + /// A column names a CAS object when it ends in artifact_object_id — which is a NAMING convention, not a + /// foreign key, so it catches a new column added in that shape and nothing else. A citer that named an object + /// under some other column name would pass this check; the pinned list above is what a reviewer reads for the + /// complete answer. Three tables are excluded, each for a stated reason rather than because it was inconvenient: + /// + /// artifact_object and artifact_location are the object's own identity and its placements, + /// not a second holder of its bytes — the cursor reads both directly, and a purge is precisely what advances a + /// location's state. artifact_transfer_intent is excluded because a saga in flight CANNOT name an object: + /// ck_artifact_transfer_intent_outcome requires the column to be null on every non-terminal row, so every + /// row that names one is a finished transfer, and a transfer is what wrote each log segment in the first place. + /// An integration test asserts that refusal rather than trusting this comment. + /// + private static bool IsObjectSoftLink(string table, string column) => + column.EndsWith("artifact_object_id", StringComparison.Ordinal) + && table is not ("artifact_object" or "artifact_location" or "artifact_cas_purge_claim" or "artifact_transfer_intent"); + + private static CodeSpaceDbContext BuildContext() => + new(new DbContextOptionsBuilder().UseNpgsql(UnreachableDatabase).UseSnakeCaseNamingConvention().Options); +} diff --git a/frontend/src/api/agentRunLogsApi.test.ts b/frontend/src/api/agentRunLogsApi.test.ts index 1cfd838ee..c20caed8f 100644 --- a/frontend/src/api/agentRunLogsApi.test.ts +++ b/frontend/src/api/agentRunLogsApi.test.ts @@ -71,6 +71,18 @@ describe("Agent Run durable log API", () => { await expect(agentsApi.readRunLogRange("r", "missing", 0, 1)).resolves.toMatchObject({ availability: "Missing", isRetryable: false }); }); + // A reclaimed archive and a lost object both answer 410. If the union or the runtime Set omits Purged, the reader + // falls through to InvalidResponse and the operator is told the RESPONSE was malformed — strictly worse than the + // wrong-but-well-formed answer, and it hides a healthy deployment doing exactly what its retention policy says. + it("keeps the retention plane's Purged verdict distinct from a missing object", async () => { + vi.stubGlobal("fetch", vi.fn() + .mockResolvedValueOnce(json({ availability: "Purged", code: "log_bytes_purged", isRetryable: false, streamId: "s" }, 410)) + .mockResolvedValueOnce(json({ availability: "PhysicalObjectMissing", code: "artifact_missing", isRetryable: false, streamId: "s" }, 410))); + + await expect(agentsApi.readRunLogRange("r", "s", 0, 1)).resolves.toEqual({ availability: "Purged", code: "log_bytes_purged", isRetryable: false }); + await expect(agentsApi.readRunLogRange("r", "s", 0, 1)).resolves.toEqual({ availability: "PhysicalObjectMissing", code: "artifact_missing", isRetryable: false }); + }); + it("fails closed when a success response omits or contradicts its range contract", async () => { vi.stubGlobal("fetch", vi.fn(() => content(new Uint8Array([1, 2]), { "X-CodeSpace-Log-Offset": "0", diff --git a/frontend/src/api/agents.ts b/frontend/src/api/agents.ts index e72d35514..00f42da5b 100644 --- a/frontend/src/api/agents.ts +++ b/frontend/src/api/agents.ts @@ -246,6 +246,8 @@ export interface AgentRunLogStreamSummary { totalBytes: number; sha256: string | null; integrity?: AgentRunLogIntegrity | null; + /** When the retention plane reclaimed this stream's bytes. Set means the head row is a tombstone: the capture settled, the window elapsed, and `integrity` is deliberately absent because there is nothing left to verify. */ + purgedAt?: string | null; createdAt: string; lastModifiedAt: string; completedAt: string | null; @@ -257,7 +259,8 @@ export interface AgentRunLogPage { nextCursor: string | null; } -export type AgentRunLogReadAvailability = "InvalidRange" | "PhysicalObjectMissing" | "IntegrityFailure" | "BackendUnavailable" | "AccessDenied" | "ProviderTimeout" | "Unsupported"; +/** `Purged` is the retention plane's own answer: the capture settled and its bytes were later reclaimed on purpose. It is NOT `PhysicalObjectMissing`, which says an object that should still be there is gone. */ +export type AgentRunLogReadAvailability = "InvalidRange" | "PhysicalObjectMissing" | "IntegrityFailure" | "BackendUnavailable" | "AccessDenied" | "ProviderTimeout" | "Unsupported" | "Purged"; export interface AgentRunLogRangeAvailable { availability: "Available"; @@ -592,7 +595,7 @@ export const agentsApi = { fetchJson(`/api/agents/stats${since ? `?since=${encodeURIComponent(since)}` : ""}`), }; -const LOG_READ_AVAILABILITIES = new Set(["InvalidRange", "PhysicalObjectMissing", "IntegrityFailure", "BackendUnavailable", "AccessDenied", "ProviderTimeout", "Unsupported"]); +const LOG_READ_AVAILABILITIES = new Set(["InvalidRange", "PhysicalObjectMissing", "IntegrityFailure", "BackendUnavailable", "AccessDenied", "ProviderTimeout", "Unsupported", "Purged"]); const EVENT_DATA_READ_AVAILABILITIES = new Set(["NotReferenced", "InvalidRange", "MetadataMissing", "PhysicalObjectMissing", "IntegrityFailure", "BackendUnavailable", "AccessDenied"]); const AGENT_EVENT_KINDS = new Set(["Queued", "Started", "AssistantMessage", "Reasoning", "PlanUpdate", "ToolCall", "CommandExecuted", "FileChanged", "TestOutput", "ApprovalRequested", "ApprovalResolved", "Warning", "Error", "FinalSummary", "Completed"]); const LOG_STATUSES = new Set(["Open", "Completed", "Truncated", "Unavailable", "Corrupt", "CaptureFailed"]); diff --git a/frontend/src/api/sessions.ts b/frontend/src/api/sessions.ts index 8fc9a107f..aac8b387f 100644 --- a/frontend/src/api/sessions.ts +++ b/frontend/src/api/sessions.ts @@ -337,7 +337,7 @@ export interface RoomArtifactVerification { } /// One agent's durable log-stream health, as the Room's projector folds it. -export type RoomAgentLogStatus = "Verified" | "Captured" | "Finalizing" | "Incomplete" | "Stalled"; +export type RoomAgentLogStatus = "Verified" | "Captured" | "Finalizing" | "Incomplete" | "Stalled" | "Purged"; /// What the sandbox actually did to one producer. `Unknown` is a real absence (nothing recorded it), never a /// confinement to render as safety — the renderer must say "posture unknown" rather than show a confined glyph. diff --git a/frontend/src/components/sessions/SessionRoomView.tsx b/frontend/src/components/sessions/SessionRoomView.tsx index d881e6c73..6b2e42ca7 100644 --- a/frontend/src/components/sessions/SessionRoomView.tsx +++ b/frontend/src/components/sessions/SessionRoomView.tsx @@ -2005,6 +2005,7 @@ const PRODUCER_LOG_LABEL: Record = { Finalizing: "logs finalizing", Incomplete: "logs incomplete", Stalled: "logs held; storage unavailable", + Purged: "logs purged; retention window elapsed", }; /** diff --git a/frontend/src/components/workflows/AgentRunLogs.tsx b/frontend/src/components/workflows/AgentRunLogs.tsx index 5094c59f3..1b31ca9d4 100644 --- a/frontend/src/components/workflows/AgentRunLogs.tsx +++ b/frontend/src/components/workflows/AgentRunLogs.tsx @@ -323,6 +323,7 @@ function ReadProblem({ problem }: { problem: AgentRunLogRangeProblem }) { const title = (() => { switch (problem.availability) { case "Missing": case "PhysicalObjectMissing": return "Stored log bytes are missing"; + case "Purged": return "Log bytes were reclaimed by retention"; case "IntegrityFailure": return "Stored log bytes are corrupt"; case "BackendUnavailable": return "Storage backend unavailable"; case "AccessDenied": return "Storage access denied";