diff --git a/docs/reports/2026-09-27-c0-corpus-v2.md b/docs/reports/2026-09-27-c0-corpus-v2.md index 9e07315..347f2b1 100644 --- a/docs/reports/2026-09-27-c0-corpus-v2.md +++ b/docs/reports/2026-09-27-c0-corpus-v2.md @@ -1,7 +1,7 @@ --- -version: "0.1.0b" +version: "0.1.1b" created_at: "2026-09-27T12:00:00+07:00,Claude,working-tree" -last_update: "2026-09-27T12:00:00+07:00,Claude" +last_update: "2026-09-27T13:00:00+07:00,Claude" status: "candidate" superseded_by: null attributes: @@ -164,16 +164,25 @@ replayable fixture. 4. Review the diff of `expected/*.json`. This diff is the behaviour change being approved. -## Observation, not changed here +## Observation, since fixed -A gate verdict for a batch with no worker receipt includes the security reason -"retrieval benchmark reported a cross-tenant leak". That reason is misleading: -there was no benchmark at all, and the retrieval dimension already reports -"retrieval benchmark is missing". The golden transcript records the current -text. Correcting it is a behaviour change for a later PR. +A gate verdict for a batch with no worker receipt used to include the security +reason "retrieval benchmark reported a cross-tenant leak". That was misleading, +because no benchmark existed at all. The reason text is now split in two: + +- no benchmark: "cross-tenant isolation is unproven: the retrieval benchmark is + missing." +- a benchmark that counted leaks: "retrieval benchmark reported N + cross-tenant leak(s)." + +Both still make the security dimension a critical `FAIL`, so the gate remains +fail-closed. The only change to the golden corpus is this reason text in +C0.4-GENESISRAG17-RECEIPTS, which appears in the gate response and again in the +Stage 17 evidence that carries the verdict. ## Change log | Version | Date | Status | Summary | Commit Hash | Agent | |---|---|---|---|---|---| | 0.1.0b | 2026-09-27 | candidate | C0.4 golden corpus registry v2: replayable fixtures, golden transcripts, runner, CI gate, Tier-4 readback split out | working-tree | Claude | +| 0.1.1b | 2026-09-27 | candidate | The gate's security reason no longer reports a leak when the benchmark is missing; one case re-baselined | working-tree | Claude | diff --git a/packages/gks-core/src/pipeline.mjs b/packages/gks-core/src/pipeline.mjs index fb809ea..52916b0 100644 --- a/packages/gks-core/src/pipeline.mjs +++ b/packages/gks-core/src/pipeline.mjs @@ -651,7 +651,11 @@ export function evaluatePipelineQuality(decision, receipt, { graphReceipt = null const securityReasons = []; if (!decision.scope?.tenantId || !decision.scope?.portfolioId || !decision.scope?.businessId) securityReasons.push("decision scope is incomplete."); - if (normalizedReceipt?.benchmark?.crossTenantLeaks !== 0) securityReasons.push("retrieval benchmark reported a cross-tenant leak."); + // Fail closed either way, but say which: no benchmark proves the absence of a + // leak, which is not the same finding as a benchmark that counted one. + const leaks = normalizedReceipt?.benchmark?.crossTenantLeaks; + if (leaks === undefined) securityReasons.push("cross-tenant isolation is unproven: the retrieval benchmark is missing."); + else if (leaks !== 0) securityReasons.push(`retrieval benchmark reported ${leaks} cross-tenant leak(s).`); const security = dimension(securityReasons.length ? "FAIL" : "PASS", securityReasons.length > 0, securityReasons); const retrievalReasons = []; diff --git a/tests/contract/pipeline-genesisrag17.test.mjs b/tests/contract/pipeline-genesisrag17.test.mjs index 711ce2e..f4bc874 100644 --- a/tests/contract/pipeline-genesisrag17.test.mjs +++ b/tests/contract/pipeline-genesisrag17.test.mjs @@ -367,12 +367,30 @@ describe("GenesisRAG17 pipeline contract", () => { const gate = await service.pipelineGate({ schemaVersion: PIPELINE_SCHEMA_VERSION, scope: batch.scope, decisionId: decision.decisionId, decisionHash: decision.decisionHash, ...worker }); expect(gate).toMatchObject({ verdict: { verdict: "FAIL", allowPublication: false, receiptHash: null } }); expect(gate.verdict.dimensions.graph.reasons).toContain("actual Tier4 graph receipt is missing."); + // No benchmark means isolation is unproven -- still a critical FAIL, but + // never reported as a leak that nothing measured. + expect(gate.verdict.dimensions.security).toEqual({ result: "FAIL", critical: true, reasons: ["cross-tenant isolation is unproven: the retrieval benchmark is missing."] }); const evidence = (await service.pipelineEvidence({ schemaVersion: PIPELINE_SCHEMA_VERSION, scope: batch.scope, runId: batch.runId, ...auth(batch.scope) })).rows; expect(evidence.map((row) => row.stageNumber)).toEqual([9, 10, 11, 12, 17]); expect(evidence.at(-1)).toMatchObject({ stageNumber: 17, outcome: "FAILED", details: { verdict: expect.objectContaining({ verdict: "FAIL" }), publicationReceipt: null } }); await expect(service.pipelinePublicationReceipt({ receipt: { schemaVersion: PIPELINE_SCHEMA_VERSION, scope: batch.scope, runId: batch.runId, decisionId: decision.decisionId, decisionHash: decision.decisionHash, snapshotId: "s", generation: "g", receiptHash: "a".repeat(64), publishedAt: "2026-09-07T15:00:03.000Z", pointerHash: "b".repeat(64), modelRevision: PIPELINE_MODEL.revision, transactionFrontier: "f", readback: { ok: true } }, ...worker })).rejects.toMatchObject({ code: "gks_conflict" }); }); + it("reports the measured count when the retrieval benchmark finds cross-tenant leaks", async () => { + const { service } = harness(); + const batch = makeBatch({ id: "batch-leaking-benchmark" }); + const { decision } = await submitAndClaim(service, batch); + const worker = auth(batch.scope, "worker"); + const graphReceipt = graphReceiptFor(decision); + const graphResult = await service.pipelineGraphReceipt({ receipt: graphReceipt, ...worker }); + const receipt = receiptFor(decision, graphResult, graphReceipt); + await service.pipelineWriteReceipt({ receipt: { ...receipt, benchmark: { ...receipt.benchmark, crossTenantLeaks: 2 } }, ...worker }); + const gate = await service.pipelineGate({ schemaVersion: PIPELINE_SCHEMA_VERSION, scope: batch.scope, decisionId: decision.decisionId, decisionHash: decision.decisionHash, ...worker }); + expect(gate.verdict).toMatchObject({ verdict: "FAIL", allowPublication: false }); + expect(gate.verdict.dimensions.security).toEqual({ result: "FAIL", critical: true, reasons: ["retrieval benchmark reported 2 cross-tenant leak(s)."] }); + expect(gate.verdict.dimensions.retrieval.reasons).toContain("cross-tenant leak count is non-zero."); + }); + it("treats WARN as a terminal failed gate and never publishes it", async () => { const { service } = harness(); const batch = makeBatch({ id: "batch-warn-gate", scope: scope({ agentId: "agent-warn" }), entries: [{ text: "Alice and Atlas", mentions: [["Alice", "warn-alice", "Person"], ["Atlas", "warn-atlas", "Product"]] }] }); diff --git a/tests/fixtures/c0-qualification/expected/C0.4-GENESISRAG17-RECEIPTS.json b/tests/fixtures/c0-qualification/expected/C0.4-GENESISRAG17-RECEIPTS.json index 7d5dfd0..efd387a 100644 --- a/tests/fixtures/c0-qualification/expected/C0.4-GENESISRAG17-RECEIPTS.json +++ b/tests/fixtures/c0-qualification/expected/C0.4-GENESISRAG17-RECEIPTS.json @@ -2971,7 +2971,7 @@ "result": "FAIL", "critical": true, "reasons": [ - "retrieval benchmark reported a cross-tenant leak." + "cross-tenant isolation is unproven: the retrieval benchmark is missing." ] }, "retrieval": { @@ -3331,7 +3331,7 @@ "result": "FAIL", "critical": true, "reasons": [ - "retrieval benchmark reported a cross-tenant leak." + "cross-tenant isolation is unproven: the retrieval benchmark is missing." ] }, "retrieval": { diff --git a/tests/fixtures/c0-qualification/registry.json b/tests/fixtures/c0-qualification/registry.json index 33dbe64..cc240f5 100644 --- a/tests/fixtures/c0-qualification/registry.json +++ b/tests/fixtures/c0-qualification/registry.json @@ -850,7 +850,7 @@ "fixture": "cases/C0.4-GENESISRAG17-RECEIPTS.json", "expectedResult": "expected/C0.4-GENESISRAG17-RECEIPTS.json", "requestSha256": "dd696274c80ab81f0a043e99e6b3351a2f0a0fced6235a223d2862421ffb4c1f", - "expectedResultSha256": "ad1f8586071c87fb06b1dde3748bb84b8583d55b4863d2eaf4dce8815a363e25", + "expectedResultSha256": "18ebf801bbb738519e1c43142f254532fc9009f55d9fa2a8e6b1992156239191", "normalizedScope": { "portfolioId": "c0-portfolio", "tenantId": "c0-tenant", diff --git a/tests/fixtures/c0-qualification/result-manifest.json b/tests/fixtures/c0-qualification/result-manifest.json index 44fbf19..2197c81 100644 --- a/tests/fixtures/c0-qualification/result-manifest.json +++ b/tests/fixtures/c0-qualification/result-manifest.json @@ -3,7 +3,7 @@ "registryVersion": "c0-qualification/v2", "fixtureId": "gks-c0.4-golden", "recordedAt": "2026-09-27", - "recordedFrom": "f50116abdc3fe606851df4d21697027e252341c7+working-tree", + "recordedFrom": "29d61802d80292b342a62de5a7fa13a2ce3da378+working-tree", "runtime": { "node": "24.19.0", "platform": "win32",