From ce8cbb1a6f1c88bce153220ab533ff8c0fb51eb5 Mon Sep 17 00:00:00 2001 From: "Mars.P" Date: Wed, 30 Sep 2026 21:40:07 +0800 Subject: [PATCH] Page the dark-streak history past cancelled runs The consecutive-dark check reads a clause's streak out of predecessor runs' own logs and steps over runs that recorded no census (cancelled by concurrency, or red before the guard). It fetched a single page of 12 completed runs, so after a burst of pushes, each cancelling the run before it, it found too few census-bearing predecessors, read a short streak, and the lane stayed green while measuring nothing. Main run 36565851194 concluded success with two UNMEASURED clauses in the middle of a 69-run dark streak. Page the run listing instead and stop as soon as every unmeasured clause is settled: its streak cut by a run that measured it, or already long enough to red the lane. Evidence is counted per clause, because a census-bearing run that never mentions a clause says nothing about it. A run listed on two pages (the boundary moves while runs finish) is read once. Paging stops at DARK_HISTORY_RUNS_CAP=60 runs. A scan that ends still unsettled now warns that the check is INCONCLUSIVE, naming how many census-bearing runs it found in how many it scanned, instead of passing a short streak off as the answer. --- .../scripts/assert-every-filter-clause-ran.sh | 133 +++++++++-- .../assert-every-filter-clause-ran.test.sh | 207 +++++++++++++++++- 2 files changed, 313 insertions(+), 27 deletions(-) diff --git a/.github/scripts/assert-every-filter-clause-ran.sh b/.github/scripts/assert-every-filter-clause-ran.sh index 6eb543e75..06727a4a9 100755 --- a/.github/scripts/assert-every-filter-clause-ran.sh +++ b/.github/scripts/assert-every-filter-clause-ran.sh @@ -36,12 +36,18 @@ set -euo pipefail # goes red. Below it the run keeps today's warning: one dark run really can be a one-off gateway fault. readonly DARK_RUNS_TO_RED=3 -# How many completed runs to page through while looking for census-bearing predecessors. A run whose job never -# reached this guard — cancelled by concurrency, or red before it — recorded no census and is therefore evidence of -# NOTHING; it is stepped over rather than counted as a reset, and this bound keeps that stepping-over from paging -# back through the whole history. +# How many completed runs one page of the history holds. A run whose job never reached this guard — cancelled by +# concurrency, or red before it — recorded no census and is therefore evidence of NOTHING; it is stepped over rather +# than counted as a reset, which is why a single page is sometimes not enough and DARK_HISTORY_RUNS_CAP exists. readonly DARK_HISTORY_RUNS_TO_SCAN=12 +# The most predecessor runs the history read scans in total, however many pages that takes. Stepping over census-less +# runs lets a burst of cancelled ones push the evidence several pages back, and this bound keeps the search from paging +# through the whole history — every run scanned is a log archive download — when the evidence is simply not there. A +# scan that reaches it with a clause still unsettled says so, INCONCLUSIVE, instead of reporting the short streak it +# found as though it were the answer. +readonly DARK_HISTORY_RUNS_CAP=60 + # Only this branch carries a streak. A branch run is a one-off by construction — it has no predecessors to be # consecutive with — so it keeps the warning and never reds on persistence. readonly DARK_STREAK_BRANCH=main @@ -127,6 +133,14 @@ fi # workflow on the streak branch and downloads each one's log archive; `unzip -p` streams it and the table rows are # read straight back out. That needs only the GITHUB_TOKEN with `actions: read` — no new secret, no extra artifact, # and no seeding period, because the runs that already went dark recorded their census the same way. +# +# The history is read PAGE BY PAGE, and only as far back as the answer needs. A run that recorded no census is stepped +# over, and a burst of pushes makes exactly those — each push cancels the run before it — so the newest page can hold +# fewer census-bearing predecessors than a streak needs to red the lane. Reading that one page alone reported the +# shortfall as a short streak and left the lane green over a clause that had measured nothing for far longer. So the +# pages keep coming until every unmeasured clause is settled — its streak cut by a run that measured it, or already +# long enough to red the lane — or DARK_HISTORY_RUNS_CAP runs have been scanned, and a scan that ends unsettled says +# INCONCLUSIVE rather than passing the short streak it found off as the answer. census_count=0 # How many predecessors were downloaded and looked at, census-bearing or not. The streak is only as trustworthy as @@ -135,6 +149,10 @@ census_count=0 history_runs_seen=0 streaks_measured=0 history_blocker="" +# Every run id the scan has already taken, space-delimited and space-bounded. The pages are read one request at a time +# while runs keep finishing, and a run that completes in between shifts every older one down a slot — so the last run +# of one page can come back as the first of the next, and a dark run read twice would count as two. +history_run_ids=" " # Every prerequisite the history read needs, named individually: "the streak could not be checked" must say WHICH # piece is missing, or it becomes the same silent nothing this guard exists to abolish. @@ -202,40 +220,87 @@ streak_of() { printf '%s' "$streak" } +# Whether a run that MEASURED the clause has already cut its streak. Nothing further back can change a streak a +# measuring run has cut, which makes this the one fact that settles a clause for good. +streak_is_broken() { + local token="$1" census + + for census in "$censuses"/*; do + [ -e "$census" ] || break + if [ "$(state_in "$census" "$token")" = MEASURED ]; then return 0; fi + done + + return 1 +} + # Stop paging the moment every unmeasured clause has had its streak broken by a run that measured it — there is # nothing left to learn, and every further page is another log archive download. any_clause_still_dark() { - local token census + local token for token in $unmeasured_tokens; do - for census in "$censuses"/*; do - [ -e "$census" ] || break - if [ "$(state_in "$census" "$token")" = MEASURED ]; then continue 2; fi - done + if ! streak_is_broken "$token"; then return 0; fi + done - return 0 + return 1 +} + +# A clause the evidence has not settled: no run that measured it has cut its streak, and the streak has not reached +# the length that reds the lane either — so reading further back could still change what the lane says about it. +clause_undecided() { + local token="$1" + + if streak_is_broken "$token"; then return 1; fi + + [ "$(streak_of "$token")" -lt "$DARK_RUNS_TO_RED" ] +} + +any_clause_undecided() { + local token + + for token in $unmeasured_tokens; do + if clause_undecided "$token"; then return 0; fi done return 1 } -collect_history() { - local workflow runs run_id archive log +# Whether another page is worth its requests: the cap has room left, and some unmeasured clause is still undecided. +history_needs_more_runs() { + [ "$history_runs_seen" -lt "$DARK_HISTORY_RUNS_CAP" ] && any_clause_undecided +} - workflow="${GITHUB_WORKFLOW_REF%%@*}" - workflow="${workflow##*/}" +# One page of this workflow's completed runs on the streak branch, newest first: one run id per line. Not +# `gh api --paginate`, which reads EVERY page — stopping at the first page that settles the answer is the whole point +# of paging by hand. +list_history_page() { + local workflow="$1" page="$2" - if ! runs="$(gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/${workflow}/runs?branch=${DARK_STREAK_BRANCH}&status=completed&per_page=${DARK_HISTORY_RUNS_TO_SCAN}" --jq '.workflow_runs[].id' 2>/dev/null)"; then - history_blocker="the workflow's run history could not be listed (is \`actions: read\` granted?)" - return 1 - fi + gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/${workflow}/runs?branch=${DARK_STREAK_BRANCH}&status=completed&per_page=${DARK_HISTORY_RUNS_TO_SCAN}&page=${page}" --jq '.workflow_runs[].id' 2>/dev/null +} + +run_already_scanned() { + case "$history_run_ids" in + *" $1 "*) return 0 ;; + esac + + return 1 +} + +# Read the census out of each run on one listed page, newest first, stopping early the moment no clause has anything +# left to learn — or the cap says enough. +scan_history_page() { + local runs="$1" run_id archive log archive="${work}/predecessor-logs.zip" log="${work}/predecessor.log" for run_id in $runs; do if [ "$run_id" = "${GITHUB_RUN_ID}" ]; then continue; fi + if run_already_scanned "$run_id"; then continue; fi + if [ "$history_runs_seen" -ge "$DARK_HISTORY_RUNS_CAP" ]; then break; fi + history_run_ids="${history_run_ids}${run_id} " history_runs_seen=$((history_runs_seen + 1)) # `unzip -p` streams every member to stdout, so the whole run's log is read in one pass without extracting a @@ -254,12 +319,35 @@ collect_history() { if ! any_clause_still_dark; then break; fi done +} + +collect_history() { + local workflow page=1 runs scanned_before + + workflow="${GITHUB_WORKFLOW_REF%%@*}" + workflow="${workflow##*/}" + + while history_needs_more_runs; do + if ! runs="$(list_history_page "$workflow" "$page")"; then + history_blocker="the workflow's run history could not be listed (is \`actions: read\` granted?)" + return 1 + fi + + scanned_before="$history_runs_seen" + scan_history_page "$runs" + + # A page that added no run the scan had not already taken is the end of the history — or an API that ignores + # `page` — so there is nothing further back to ask for. + if [ "$history_runs_seen" -eq "$scanned_before" ]; then break; fi + + page=$((page + 1)) + done # No predecessor recorded a census at all — the workflow's first run on this branch, a retention gap, or a token # that can list runs but not read their logs. Whatever the cause, there is nothing to be consecutive WITH, so say # so rather than reporting a confident streak of one. if [ "$census_count" -eq 0 ]; then - history_blocker="no completed ${DARK_STREAK_BRANCH} run carried a clause census to compare against" + history_blocker="INCONCLUSIVE: 0 of the ${history_runs_seen} completed ${DARK_STREAK_BRANCH} runs scanned carried a clause census to compare against" return 1 fi @@ -286,6 +374,13 @@ if [ -n "$unmeasured_tokens" ] && [ "$streaks_measured" -ne 1 ] && [ "${GITHUB_R echo "::warning::…and the consecutive-dark check could NOT run (${history_blocker}), so a clause that has measured nothing for runs on end still reads here as a one-off." fi +# A scan that ended — the cap reached, or the history itself — with a clause still unsettled has not found a short +# streak, it has found no answer. Reporting it as the one-off it looks like is how a burst of cancelled runs used to +# read green, so say what the verdict rests on instead. +if [ "$streaks_measured" -eq 1 ] && any_clause_undecided; then + echo "::warning::…and the consecutive-dark check is INCONCLUSIVE: only ${census_count} of the ${history_runs_seen} predecessor runs scanned carried a census, and too few of those recorded the clause to settle its streak (the scan stops at ${DARK_HISTORY_RUNS_CAP} runs), so a clause that has measured nothing for runs on end may still read here as a one-off." +fi + # The skip reason RealModelGate recorded for this clause's first skipped test. The streak says the instrument is # dark; this says why it went dark, which is the half an operator can act on. skip_reason() { diff --git a/.github/scripts/assert-every-filter-clause-ran.test.sh b/.github/scripts/assert-every-filter-clause-ran.test.sh index 7ef342c9c..4fa12a2fd 100755 --- a/.github/scripts/assert-every-filter-clause-ran.test.sh +++ b/.github/scripts/assert-every-filter-clause-ran.test.sh @@ -172,16 +172,33 @@ fi history="${tmp}/history" stub_bin="${tmp}/stub-bin" summary="${tmp}/step-summary.md" +gh_requests="${tmp}/gh-requests.log" mkdir -p "$history" "$stub_bin" -# The `gh` stub answers the only two calls the guard makes — list this workflow's completed runs, and download one -# run's log archive. Stubbing the CLI rather than the guard's own lookup keeps the guard's real endpoints, its real -# `unzip -p` streaming and its real census parser under test; only the network is replaced. +# The `gh` stub answers the only two calls the guard makes — list ONE PAGE of this workflow's completed runs, and +# download one run's log archive. Stubbing the CLI rather than the guard's own lookup keeps the guard's real endpoints, +# its real `unzip -p` streaming and its real census parser under test; only the network is replaced. Every request is +# appended to a log, because "the guard stopped paging" is visible nowhere else. cat > "${stub_bin}/gh" <<'STUB' #!/usr/bin/env bash endpoint="$2" +printf '%s\n' "$endpoint" >> "${GUARD_TEST_GH_LOG:-/dev/null}" + case "$endpoint" in - *"/runs?branch="*) cat "${GUARD_TEST_HISTORY}/run-ids" ;; + *"/runs?branch="*) + page=1 + case "$endpoint" in *"&page="*) page="${endpoint##*&page=}" ;; esac + + # A guard that keeps asking for pages once the history has run out would loop forever; fail the listing instead, + # so that regression fails the suite rather than hanging it. No case here stages more than 6 pages. + if [ "$(grep -c 'page=' "${GUARD_TEST_GH_LOG:-/dev/null}")" -gt 8 ]; then exit 1; fi + + # A page nobody staged is past the end of the history, which the API answers with an empty page — except page 1, + # where a missing listing stands for the listing call itself failing. + if [ -f "${GUARD_TEST_HISTORY}/run-ids.${page}" ]; then exec cat "${GUARD_TEST_HISTORY}/run-ids.${page}"; fi + if [ "$page" -gt 1 ]; then exit 0; fi + exit 1 + ;; */logs) run_id="${endpoint%/logs}"; exec cat "${GUARD_TEST_HISTORY}/${run_id##*/}.zip" ;; *) exit 1 ;; esac @@ -238,7 +255,9 @@ make_censusless_predecessor() { (cd "$dir" && zip -qq "${history}/${run_id}.zip" "0_real model (a lane).txt") } -set_history() { printf '%s\n' "$@" > "${history}/run-ids"; } +# A run listing is one file per page, `run-ids.`; set_history stages page 1, the only page most cases ever need. +set_history_page() { local page="$1"; shift; printf '%s\n' "$@" > "${history}/run-ids.${page}"; } +set_history() { set_history_page 1 "$@"; } # The guard as GitHub runs it on the streak branch: run 999 is THIS run and must be skipped in its own history. # The history read's two fallible prerequisites — the token and the listing endpoint — are parameters, because the @@ -247,8 +266,10 @@ run_on_history() { local token="$1" history_dir="$2" ref="$3"; shift 3 : > "$summary" + : > "$gh_requests" env PATH="${stub_bin}:${PATH}" \ GUARD_TEST_HISTORY="$history_dir" \ + GUARD_TEST_GH_LOG="$gh_requests" \ GH_TOKEN="$token" \ GITHUB_TOKEN="$token" \ GITHUB_REF="$ref" \ @@ -274,17 +295,36 @@ expect_summary() { fi } +# Which pages of run history the guard asked for, in order. Where paging STOPPED is visible nowhere else: reading page 3 +# as well can end in exactly the same verdict, and only the requests tell the two apart. +expect_pages() { + local want="$1" name="$2" got; shift 2 + "$@" >/dev/null 2>&1 + got="$(sed -nE 's/.*[?&]page=([0-9]+).*/\1/p' "$gh_requests" | tr '\n' ' ' | sed 's/ $//')" + + if [ "$got" = "$want" ]; then + echo " ok ${name}" + else + echo " FAILED ${name} — the guard asked for history pages '${got}', expected '${want}'" + failures=$((failures + 1)) + fi +} + # Rule 8: the threshold is a named constant, changed by a PR. Pinned literally, because moving it silently changes # how long a gate may report green over an instrument that never ran. expect_output has "readonly DARK_RUNS_TO_RED=3" "the dark-run threshold is pinned at 3" \ grep -F "readonly DARK_RUNS_TO_RED=3" "$guard" -# The other two knobs decide the same thing from the other side: how far back evidence is looked for, and whose -# history counts as a streak at all. Raising the scan bound silently changes how many censusless runs can be stepped -# over; changing the branch silently turns the streak off everywhere. Pinned for the same reason as the threshold. +# The other knobs decide the same thing from the other side: how many runs one page of history holds, how many runs the +# pages may add up to, and whose history counts as a streak at all. Raising the cap silently changes how many censusless +# runs can be stepped over — and how many log archives one guard run downloads; changing the branch silently turns the +# streak off everywhere. Pinned for the same reason as the threshold. expect_output has "readonly DARK_HISTORY_RUNS_TO_SCAN=12" "the history scan bound is pinned at 12" \ grep -F "readonly DARK_HISTORY_RUNS_TO_SCAN=12" "$guard" +expect_output has "readonly DARK_HISTORY_RUNS_CAP=60" "the history cap is pinned at 60" \ + grep -F "readonly DARK_HISTORY_RUNS_CAP=60" "$guard" + expect_output has "readonly DARK_STREAK_BRANCH=main" "the streak branch is pinned to main" \ grep -F "readonly DARK_STREAK_BRANCH=main" "$guard" @@ -412,6 +452,157 @@ expect_summary '| `RealModelBenchmark` | UNMEASURED | 0 | 0 | 1 | 1 |' \ "a MISSING predecessor resets the streak to 1 rather than stepping over it" \ run_on refs/heads/main "$dark_trx" RealModelBenchmark +# ── Paging: the evidence for a streak can sit several pages back ────────────────────────────────────────────────── +# +# A predecessor with no census is stepped over, and a burst of pushes makes exactly those — each push cancels the run +# before it — so the newest page of history can hold fewer census-bearing runs than a streak needs. Read as ONE page, +# that shortfall looked like a short streak: a real main run went green over two UNMEASURED clauses in the middle of a +# 69-run dark streak. The history below is staged as pages of 12 (runs 101..112 are page 1, 201..212 page 2, and so +# on) with every run starting out as a cancelled, census-less one; each case turns only the few it needs into +# census-bearing ones. + +# `pages` full pages of 12 census-less runs, and no page beyond them. +reset_paged_history() { + local pages="$1" page run_id + + rm -f "${history}"/run-ids.* + make_censusless_predecessor 100 + + for page in $(seq 1 "$pages"); do + set_history_page "$page" $(seq "${page}01" "${page}12") + + for run_id in $(seq "${page}01" "${page}12"); do cp "${history}/100.zip" "${history}/${run_id}.zip"; done + done +} + +# THE regression. Page 1 holds one dark census among eleven cancelled runs, page 2 two more dark ones. One page alone +# read a streak of 2 and stayed green; paged, the guard finds three dark predecessors — a streak of 4 with this run — +# and reds. +reset_paged_history 5 +make_predecessor 105 "$dark_trx" RealModelBenchmark +make_predecessor 203 "$dark_trx" RealModelBenchmark +make_predecessor 209 "$dark_trx" RealModelBenchmark + +expect 1 "a streak whose evidence sits on page 2 REDS the lane — one page alone read streak 2 and stayed green" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_output has "::error::RealModelBenchmark has now measured NOTHING on 4 consecutive main runs" \ + "the paged streak counts the dark runs of every page it read" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_output has "3 of the 24 predecessor runs scanned carried a census" \ + "the error's evidence base is the whole paged scan, not its first page" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +# ...and it read exactly as far as it needed: pages 3-5 are staged and never asked for. +expect_pages "1 2" "paging stops at the page that settles the streak" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +# A run that MEASURED the clause settles it alone. Page 1 holds one dark census, page 2 a measuring run: the streak is +# cut at 2 and nothing older can change that, so page 3 — staged with two dark runs — must never be asked for. +reset_paged_history 5 +make_predecessor 104 "$dark_trx" RealModelBenchmark +make_predecessor 202 "$measured_trx" RealModelBenchmark +make_predecessor 301 "$dark_trx" RealModelBenchmark +make_predecessor 302 "$dark_trx" RealModelBenchmark + +expect 0 "a measuring run on page 2 cuts the streak: the lane only warns" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_pages "1 2" "a measuring run on page 2 stops the paging — page 3 is never asked for" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_summary '| `RealModelBenchmark` | UNMEASURED | 0 | 0 | 1 | 2 |' \ + "the streak it stops at is the one the evidence shows" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_summary "Streaks read from 2 of the 14 predecessor runs scanned" \ + "it stops reading at the run that settled the streak, mid-page" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_output lacks "INCONCLUSIVE" "a settled streak is never reported as inconclusive" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +# Nothing settles it. One dark census on page 1 and, on page 3, a census-bearing run whose lane never reported this +# clause, then silence: the scan reads exactly DARK_HISTORY_RUNS_CAP runs — page 6, staged with the two dark runs that +# WOULD red the lane, is never read — and says so, rather than passing the streak of 2 it found off as the whole story. +# Still not red: a run that said nothing is not a dark run. +reset_paged_history 5 +make_predecessor 106 "$dark_trx" RealModelBenchmark +make_predecessor 302 "$trx" RealModelSupervisor +make_predecessor 601 "$dark_trx" RealModelBenchmark +make_predecessor 602 "$dark_trx" RealModelBenchmark +set_history_page 6 601 602 + +expect 0 "a scan that reaches the cap unsettled warns instead of redding the lane" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_output has "is INCONCLUSIVE: only 2 of the 60 predecessor runs scanned carried a census" \ + "an unsettled scan says INCONCLUSIVE, with how many census-bearing runs it found in how many it scanned" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_summary "Streaks read from 2 of the 60 predecessor runs scanned" \ + "the step summary names the same paged evidence base" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_pages "1 2 3 4 5" "the cap is hard: nothing beyond DARK_HISTORY_RUNS_CAP runs is requested" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +# A run that finishes while the pages are being read shifts every older run down a slot, so the LAST run of page 1 comes +# back as the FIRST of page 2. A dark run read twice must still count once: here the double count would turn a streak +# of 2 into a red. +reset_paged_history 2 +make_predecessor 112 "$dark_trx" RealModelBenchmark +set_history_page 2 112 201 202 203 204 205 206 207 208 209 210 211 + +expect 0 "a run listed on two pages counts once" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_summary "Streaks read from 1 of the 23 predecessor runs scanned" "a run listed on two pages is scanned once" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +# The cap counts RUNS, not pages. That same shifted boundary makes page 2 add only 11 new runs, so five pages come to 59 +# and the 60th run is the first of page 6. The scan must take that one run and stop — not the rest of the page, whose +# second run is another dark one that would make the streak 4 over 61 runs. +reset_paged_history 5 +make_predecessor 106 "$dark_trx" RealModelBenchmark +make_predecessor 601 "$dark_trx" RealModelBenchmark +make_predecessor 602 "$dark_trx" RealModelBenchmark +set_history_page 2 112 201 202 203 204 205 206 207 208 209 210 211 +set_history_page 6 601 602 + +expect_output has "reds at 3; 2 of the 60 predecessor runs scanned carried a census" \ + "the cap counts runs, not pages: the scan stops at the 60th run, mid-page" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +# A census that never mentions the clause — another lane's tables from a run this lane was cancelled in — is evidence of +# nothing for THIS clause, so it cannot count toward "read enough" either. Page 1 holds three census-bearing runs but +# only one dark reading of the clause: counting census-bearing runs would stop there at a streak of 2 and stay green. +# The guard has to read on to page 2, where the second dark reading makes three. +reset_paged_history 5 +make_predecessor 101 "$dark_trx" RealModelBenchmark +make_predecessor 102 "$trx" RealModelSupervisor +make_predecessor 103 "$trx" RealModelSupervisor +make_predecessor 201 "$dark_trx" RealModelBenchmark + +expect 1 "census-bearing runs that never mention the clause are not enough evidence to stop paging" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_pages "1 2" "...and the paging still stops once the clause's own streak is long enough" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +# A history where no run carried a census (every one cancelled before the guard) gives a streak nothing to be +# consecutive with. That already warned; now it must say how many runs it looked through, or "found no census" reads +# the same after 12 runs as after 60. +reset_paged_history 1 + +expect 0 "a history with no census at all warns instead of redding the lane" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + +expect_output has "could NOT run (INCONCLUSIVE: 0 of the 12 completed main runs scanned carried a clause census" \ + "the no-census warning names how many runs it looked through" \ + run_on refs/heads/main "$dark_trx" RealModelBenchmark + if [ "$failures" -ne 0 ]; then echo "${failures} guard self-test(s) failed" exit 1