From b264790fb3c5ca232155bbbde18c572f2eff8ccb Mon Sep 17 00:00:00 2001
From: rustnew
Date: Thu, 3 Sep 2026 14:03:43 +0100
Subject: [PATCH] Test PROBE-mode tie-break for the Xavier/He blind spot
(negative result)
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Third fix attempt for zc_jacobcov's proven blind spot (results/reports/
2026-09-02T08-04-49Z_explore_scale_invariance_blindspot.md): jacob_cov's
binary activation-sign statistic is exactly invariant to the positive
rescaling that separates Xavier from He, so no PURE-mode secondary proxy
can ever break that tie -- both static attempts already tried
(TieBreakHeuristicPredictor's raw and population-normalized gradient_norm
variants) failed for that exact reason.
Adds precog.meta_predictor.ProbeTieBreakPredictor: when jacob_cov ties,
spend a real, bounded PROBE-mode budget (docs.md §5, 50 steps -- the
cheapest the Zero-Training Contract allows) on just the tied candidates,
and pick whichever ends with the lower loss. Tested in
scripts/explore_probe_tiebreak.py on the same locked TEST split used
throughout this project.
Result: he-recall rises from 0% to 30% (3/10 tasks where "he" is truly
best), but overall regret gets *worse* (+14.6 -> +45.4 steps) at a cost
of 68 extra steps/decision (37.5% of a mean FULL TRAINING run) -- 50
steps is enough to make "he" look locally better, not enough to see it
sometimes never converges at all within budget. Net: not a win, kept and
reported as a negative result like every other failed attempt in this
project (LSUV init, active sampling).
Added to CI (reproduce.yml). docs/index.html's "bug we found but can't
fix" section updated from "two attempted fixes" to "three", with a link
to the new report. data/meta_dataset.db and results/gate_evaluations.csv
change because record_gate_evaluation() logs this run's single new gate
row (docs.md §12 meta-dataset) -- a real, explained addition, not a
side-effect of running scripts locally before committing.
Co-Authored-By: Claude Sonnet 5
Claude-Session: https://claude.ai/code/session_01WSc9sb1otU6ssfxDBzeNHG
---
.github/workflows/reproduce.yml | 3 +
data/meta_dataset.db | Bin 2568192 -> 2568192 bytes
docs/index.html | 13 +-
precog/meta_predictor.py | 87 ++++++
results/gate_evaluations.csv | 1 +
...-09-03T12-59-38Z_explore_probe_tiebreak.md | 128 +++++++++
scripts/explore_probe_tiebreak.py | 251 ++++++++++++++++++
7 files changed, 481 insertions(+), 2 deletions(-)
create mode 100644 results/reports/2026-09-03T12-59-38Z_explore_probe_tiebreak.md
create mode 100644 scripts/explore_probe_tiebreak.py
diff --git a/.github/workflows/reproduce.yml b/.github/workflows/reproduce.yml
index 55f3c66..3b551c9 100644
--- a/.github/workflows/reproduce.yml
+++ b/.github/workflows/reproduce.yml
@@ -45,3 +45,6 @@ jobs:
- name: LSUV data-aware init (4th candidate, underperformed)
run: python scripts/explore_lsuv_init.py
+
+ - name: PROBE-mode tie-break for the Xavier/He blind spot (raises he-recall, nets worse regret)
+ run: python scripts/explore_probe_tiebreak.py
diff --git a/data/meta_dataset.db b/data/meta_dataset.db
index 6b4d7766588643190c03212966a0a27896bf0530..b0f2fc9aeeee0026cc742c9931fa252a712b4304 100644
GIT binary patch
delta 455
zcmZ9`%TB^T7zJQzfpW8mcLlZHNK7a#G$4>Pn(!cFse@FIHf_bFpbNLS_Z^Heh8O6H
ztq&k;mp*~VA}*Zd+noQ*KhvK)F?|@Onek($&WvBqbcy-A_oN3;Vqgwf2to+Lz(E9}
z5Q8`*U>!DK6Vi}@E!c(~5MUSfU>^?PaO6qZ)1kn#M}Z{A
zSLf@AUc4WBj=n`cxCs{yvmqgf;cLnYv9|+laG3}wQdH$4QRYr8%EhuMS4CM73dLHf
zT2m^$f@s-hgNm*}8#dM2ElO;vYn=}9b27_M-WC7EZ0&}i)G}O0m8F856K0ivvDN%p
z{1+;Ice>PZspY68+`Xb&_xB~;bX+o>lq-29Czw5Ix2SebEZrq$lla}{HNNKBn$b17
P7piicFO7IF`_=vi(%GJz
delta 227
zcmWN=yAHun0Dxgft4B*u-LF-3>)P2S>2pXVMvKYZ#9&0M25%q+=?k32Q?Ob+f&WXs
z^m7#{Pkr$^`-ylxR_j=N_c}i4STIk6CJVGES)@&eC6?(@(PM>G)>vnQO}5x(hh6sA
zr_TWc4jFR9F(;gImgzWJoC4QWIZvv&o0NB@)VbTw;#+>CWf_IgR9W>Vm&PUB9y6t*
G-{ucH{8K~#
diff --git a/docs/index.html b/docs/index.html
index bb261f7..bcd4744 100644
--- a/docs/index.html
+++ b/docs/index.html
@@ -246,10 +246,19 @@ 04A bug we found but can't fix
exactly, for the same seed — checked across all 312 tasks, max difference
is 0.0. Two other proxies (effective_rank,
jacobian_condition_mean) share the property for the same
- reason. Two attempted fixes (raw and population-normalized tie-breaking
- with gradient_norm) both failed.
+ reason. Three attempted fixes, all failed: raw and population-normalized
+ tie-breaking with gradient_norm (same structural blindness,
+ different proxy), and — the one thing that should have worked, since
+ it stops reading a PURE-mode proxy entirely — a bounded 50-step
+ PROBE-mode
+ run on just the tied candidates. It does recover some "he" picks
+ (recall 0%→30% on the 10 tasks where "he" is truly best), but net
+ regret gets worse (+14.6→+45.4 steps): 50 real training steps
+ is enough to make "he" look locally better, not enough to see that it
+ sometimes never converges at all within budget.
Full audit of all 11 proxies →
+ Why the PROBE-mode fix doesn't net out →
diff --git a/precog/meta_predictor.py b/precog/meta_predictor.py
index 2e986c1..cb78b85 100644
--- a/precog/meta_predictor.py
+++ b/precog/meta_predictor.py
@@ -52,6 +52,7 @@
from precog.meta_knowledge_base import MetaKnowledgeBase
from precog.model import InitMethod
+from precog.modes import Mode, TrainingConfig, TrainProtocol, train
from precog.regime import _bucket_noise, _bucket_volume
from precog.trainability import zero_cost_features
@@ -151,6 +152,7 @@ class Recommendation:
steps_range: tuple[float, float] # +/- 1 std across the ensemble
confidence: float # 1 - (relative spread), clamped to [0, 1]
per_candidate: dict[str, dict] # every candidate's own prediction, for transparency
+ probe_cost_steps: int = 0 # PROBE-mode budget actually spent on this decision (docs.md §5 cost-accounting)
class MetaPredictor:
@@ -433,6 +435,91 @@ def secondary_key(k: str) -> float:
)
+class ProbeTieBreakPredictor:
+ """Third fix attempt for zc_jacobcov's proven blind spot (see
+ TieBreakHeuristicPredictor above): jacob_cov's binary activation-sign
+ statistic is *exactly* invariant to the positive rescaling that
+ separates Xavier from He (max |xavier-he| jacob_cov = 0.0 across all
+ 312 meta-dataset tasks), so no PURE-mode secondary proxy -- raw or
+ population-normalized gradient_norm, both tried in
+ scripts/compare_meta_predictors.py -- can ever break that exact tie;
+ gradient_norm turned out to carry the same he>xavier scale confound
+ jacob_cov's sign-only statistic doesn't even look at.
+
+ This tries the other option named in
+ results/reports/2026-09-02T08-04-49Z_explore_scale_invariance_blindspot.md:
+ a minimal PROBE-mode check (docs.md §5: DeltaW != 0, but bounded and
+ logged, 50-1000 steps by contract) spent *only* on the exact tie
+ jacob_cov cannot see -- train each tied candidate for `probe_steps`
+ real steps at its own (learning_rate, batch_size, optimizer) and keep
+ whichever ends with the lower loss. `last_probe_cost_steps` records
+ the budget actually spent on the most recent call, so callers can
+ report it per the Zero-Training Contract's own requirement ("must
+ always be possible to answer how much PROBE adds over PURE alone, for
+ what additional cost") -- see scripts/explore_probe_tiebreak.py."""
+
+ def __init__(
+ self,
+ primary_proxy: str = "jacob_cov",
+ primary_higher_is_better: bool = False,
+ tie_tolerance: float = 1e-6,
+ probe_steps: int = 50,
+ ):
+ self.primary_proxy = primary_proxy
+ self.primary_higher_is_better = primary_higher_is_better
+ self.tie_tolerance = tie_tolerance
+ self.probe_steps = probe_steps
+ self.last_probe_cost_steps = 0
+
+ def recommend(
+ self,
+ features_row: pd.DataFrame,
+ zero_cost_by_candidate: dict[InitMethod, dict],
+ architecture=None,
+ x: torch.Tensor | None = None,
+ y: torch.Tensor | None = None,
+ training_by_candidate: dict[InitMethod, TrainingConfig] | None = None,
+ ) -> Recommendation:
+ per_candidate = {
+ c.value: {"expected_steps": float("nan"), "std_steps": 0.0, "primary_score": zc[self.primary_proxy]}
+ for c, zc in zero_cost_by_candidate.items()
+ }
+ primary_sign = -1 if self.primary_higher_is_better else 1
+ primary_values = {k: v["primary_score"] * primary_sign for k, v in per_candidate.items()}
+ best_primary = min(primary_values.values())
+ tied = [k for k, v in primary_values.items() if abs(v - best_primary) <= self.tie_tolerance]
+
+ self.last_probe_cost_steps = 0
+ if len(tied) == 1:
+ best_init_name = tied[0]
+ else:
+ if architecture is None or x is None or y is None or training_by_candidate is None:
+ raise ValueError(
+ "ProbeTieBreakPredictor needs a live architecture/x/y/training_by_candidate "
+ "context to actually run the PROBE that breaks the tie -- pass them through, "
+ "see scripts/explore_probe_tiebreak.py for how the harness wires this up."
+ )
+ probe_losses = {}
+ for k in tied:
+ training = training_by_candidate[InitMethod(k)]
+ protocol = TrainProtocol(
+ mode=Mode.PROBE, max_steps=self.probe_steps, loss_threshold=-1.0, seed=0
+ )
+ result = train(architecture, x, y, training, protocol)
+ probe_losses[k] = result.final_loss
+ self.last_probe_cost_steps += self.probe_steps
+ best_init_name = min(probe_losses, key=probe_losses.get)
+
+ return Recommendation(
+ recommended_init=InitMethod(best_init_name),
+ expected_steps=float("nan"),
+ steps_range=(float("nan"), float("nan")),
+ confidence=float("nan"), # this method makes no probabilistic claim -- see docs.md §20
+ per_candidate=per_candidate,
+ probe_cost_steps=self.last_probe_cost_steps,
+ )
+
+
class KNNMetaPredictor:
"""Alternative to the RandomForest MetaPredictor (docs.md §19 ablation
spirit): predicts purely from the Meta-Knowledge Base's (§9.6) nearest
diff --git a/results/gate_evaluations.csv b/results/gate_evaluations.csv
index c4b4646..4e655fc 100644
--- a/results/gate_evaluations.csv
+++ b/results/gate_evaluations.csv
@@ -232,3 +232,4 @@ evaluation_id,timestamp,generation,gate_number,metric_name,metric_value,threshol
231,2026-09-02 07:46:49,v1-meta-predictor-zc_jacobcov_normtiebreak,2,mean_regret_steps_to_threshold,47.666666666666664,0.0,1,60,regret = steps(predicted_init) - steps(true_best_init); relative_regret=+33.72%; universal_baseline_regret=+22.2; random_baseline_regret=+43.2
232,2026-09-02 08:04:49,v1-scale-invariance-blindspot,0,fraction_proxies_blind_to_xavier_vs_he,0.2727272727272727,0.0,1,312,"blind_proxies=['jacob_cov', 'effective_rank', 'jacobian_condition_mean'], tie_threshold=5% mean relative difference"
233,2026-09-02 19:28:18,v1-trainability-engine-at-scale,1,spearman_rho_gradient_norm_vs_steps_full_scale,0.5398950415558994,0.7,0,936,"full meta-dataset re-check (312 tasks) of gate1_ranking.py's original n=36 (12-task) result; no new training runs, same controlled design (§21)"
+234,2026-09-03 12:59:38,v1-probe-tiebreak,1,he_recall_probe_tiebreak,0.3,0.0,1,10,"raw_he_hits=0/10, tiebreak_he_hits=0/10, probe_he_hits=3/10, probe_steps=50, mean_probe_cost_steps=68.3, overhead_pct_of_mean_full_training=37.5"
diff --git a/results/reports/2026-09-03T12-59-38Z_explore_probe_tiebreak.md b/results/reports/2026-09-03T12-59-38Z_explore_probe_tiebreak.md
new file mode 100644
index 0000000..96f4486
--- /dev/null
+++ b/results/reports/2026-09-03T12-59-38Z_explore_probe_tiebreak.md
@@ -0,0 +1,128 @@
+# PROBE-Mode Tie-Break for the Xavier/He Blind Spot
+
+_Generated 2026-09-03T12-59-38Z (UTC)_
+
+## Method
+
+Three candidates evaluated once each on the identical locked TEST split
+(180 rows, 60 tasks) used throughout this
+project:
+
+- `zc_jacobcov` -- the raw heuristic, already known to never recommend "he"
+ (0/10 on tasks where "he" is truly best).
+- `zc_jacobcov_tiebreak` -- the first static-secondary-proxy fix attempt
+ (gradient_norm), from scripts/compare_meta_predictors.py.
+- `zc_jacobcov_probetiebreak` -- this run's new candidate
+ (precog.meta_predictor.ProbeTieBreakPredictor): on the exact tie
+ jacob_cov cannot break, spends a real, bounded PROBE-mode budget
+ (50 steps per tied candidate, docs.md §5) and picks whichever
+ ends with the lower loss, instead of another PURE-mode secondary proxy.
+
+Per the Zero-Training Contract (docs.md §5: "must always be possible to
+answer how much PROBE adds over PURE alone, for what additional cost"),
+the table below reports that cost -- mean extra training steps spent per
+decision -- next to accuracy, he-recall (of the 10
+test tasks where "he" is genuinely the fastest choice) and regret.
+
+## Results
+
+| candidate | accuracy | he-recall | mean regret (steps) | mean probe cost (steps) |
+|---|---:|---:|---:|---:|
+| zc_jacobcov | 47% (28/60) | 0% (0/10) | +14.6 | 0.0 |
+| zc_jacobcov_tiebreak | 47% (28/60) | 0% (0/10) | +14.6 | 0.0 |
+| zc_jacobcov_probetiebreak | 43% (26/60) | 30% (3/10) | +45.4 | 68.3 |
+
+PROBE overhead: 68 steps/decision on average,
+against a mean 182-step FULL TRAINING run on this
+split (37.5% of it). Since jacob_cov ties on every
+single test task (the blind spot is structural, not occasional), this is
+also the candidate's *total* added cost -- there is no untied case to
+amortize it against.
+
+| seed | true best init | predicted (probetiebreak) | regret (steps) | probe cost (steps) |
+|---|---|---|---:|---:|
+| 100 | xavier | orthogonal | 10 | 0 |
+| 107 | orthogonal | orthogonal | 0 | 0 |
+| 120 | orthogonal | he | 149 | 100 |
+| 131 | xavier | xavier | 0 | 100 |
+| 132 | xavier | xavier | 0 | 100 |
+| 137 | orthogonal | xavier | 7 | 100 |
+| 141 | orthogonal | he | 128 | 100 |
+| 146 | orthogonal | orthogonal | 0 | 0 |
+| 147 | xavier | he | 973 | 100 |
+| 148 | he | orthogonal | 1 | 0 |
+| 150 | he | he | 0 | 100 |
+| 151 | orthogonal | orthogonal | 0 | 0 |
+| 155 | orthogonal | xavier | 54 | 100 |
+| 171 | orthogonal | orthogonal | 0 | 0 |
+| 172 | he | he | 0 | 100 |
+| 175 | orthogonal | xavier | 3 | 100 |
+| 197 | xavier | xavier | 0 | 100 |
+| 204 | xavier | he | 281 | 100 |
+| 211 | orthogonal | xavier | 1 | 100 |
+| 213 | xavier | xavier | 0 | 100 |
+| 222 | orthogonal | orthogonal | 0 | 0 |
+| 224 | xavier | xavier | 0 | 100 |
+| 228 | xavier | xavier | 0 | 100 |
+| 232 | he | xavier | 8 | 100 |
+| 233 | orthogonal | xavier | 71 | 100 |
+| 244 | xavier | he | 9 | 100 |
+| 249 | he | orthogonal | 55 | 0 |
+| 254 | orthogonal | orthogonal | 0 | 0 |
+| 255 | orthogonal | orthogonal | 0 | 0 |
+| 258 | xavier | orthogonal | 7 | 0 |
+| 261 | xavier | xavier | 0 | 100 |
+| 263 | orthogonal | xavier | 66 | 100 |
+| 266 | orthogonal | he | 49 | 100 |
+| 269 | orthogonal | xavier | 9 | 100 |
+| 270 | he | xavier | 8 | 100 |
+| 281 | he | xavier | 9 | 100 |
+| 283 | orthogonal | orthogonal | 0 | 0 |
+| 297 | orthogonal | xavier | 46 | 100 |
+| 304 | orthogonal | orthogonal | 0 | 0 |
+| 307 | orthogonal | xavier | 8 | 100 |
+| 315 | xavier | xavier | 0 | 100 |
+| 322 | orthogonal | xavier | 1 | 100 |
+| 326 | xavier | xavier | 0 | 100 |
+| 329 | xavier | xavier | 0 | 100 |
+| 341 | xavier | xavier | 0 | 100 |
+| 344 | orthogonal | xavier | 131 | 100 |
+| 348 | he | orthogonal | 18 | 0 |
+| 350 | orthogonal | xavier | 218 | 100 |
+| 352 | xavier | xavier | 0 | 100 |
+| 358 | he | orthogonal | 8 | 0 |
+| 360 | he | he | 0 | 100 |
+| 361 | orthogonal | he | 9 | 100 |
+| 366 | xavier | orthogonal | 7 | 0 |
+| 372 | xavier | orthogonal | 3 | 0 |
+| 378 | orthogonal | orthogonal | 0 | 0 |
+| 380 | xavier | xavier | 0 | 100 |
+| 382 | xavier | he | 5 | 100 |
+| 386 | xavier | orthogonal | 0 | 0 |
+| 390 | xavier | he | 214 | 100 |
+| 398 | orthogonal | he | 156 | 100 |
+
+## Verdict
+
+He-recall rises from
+0% (0/10) to
+30% (3/10) with the PROBE
+tie-break -- unlike the two static secondary-proxy attempts, spending real
+(bounded) training budget on the exact tie can see the Xavier/He difference
+at all, because it is the only one of the three that isn't a
+positive-rescaling-invariant statistic by construction.
+
+But regret -- this project's primary metric, precisely because it catches
+what accuracy alone can't (scripts/compare_meta_predictors.py) -- gets
+**worse**, not better: +14.6 steps (raw) to
++45.4 steps (PROBE tie-break), on top of
+68 extra steps/decision spent
+(37.5% of a mean FULL TRAINING run) to get there.
+50 steps is enough to sometimes make "he" look locally better
+than Xavier, but not enough to foresee the cases named in this project's
+own README §1 where an init that looks fine early on never actually
+converges within budget: the single worst case here is seed 147
+(true best is xavier at 627 steps, the
+probe picks "he", which needs 1600 steps to
+reach threshold -- regret +973). Net: **NOT a net improvement**
+-- it trades a narrow, cosmetic fix (jacob_cov's 'never recommends he' symptom) for a worse net decision rule, the same shape of finding as this project's other ideas that looked reasonable but underperformed a simpler baseline (LSUV init, active sampling) -- see source.md pillar 4 and §5 above.
diff --git a/scripts/explore_probe_tiebreak.py b/scripts/explore_probe_tiebreak.py
new file mode 100644
index 0000000..e0b5b22
--- /dev/null
+++ b/scripts/explore_probe_tiebreak.py
@@ -0,0 +1,251 @@
+#!/usr/bin/env python3
+"""Tests the second fix option named in
+results/reports/2026-09-02T08-04-49Z_explore_scale_invariance_blindspot.md
+for zc_jacobcov's proven blind spot: jacob_cov's binary activation-sign
+statistic is *exactly* invariant to the positive rescaling that separates
+Xavier from He (max |xavier-he| jacob_cov = 0.0 across all 312 meta-dataset
+tasks), so no PURE-mode reading of it, however combined with a secondary
+proxy, can ever tell them apart. Both static tie-breaks already tried in
+scripts/compare_meta_predictors.py (TieBreakHeuristicPredictor's raw and
+population-normalized gradient_norm variants) failed for exactly this
+reason -- gradient_norm turned out to carry the same he>xavier scale
+confound jacob_cov's sign-only statistic doesn't even try to see.
+
+precog.meta_predictor.ProbeTieBreakPredictor tries the other option the
+report named: a minimal, explicitly-costed PROBE-mode run (docs.md §5:
+DeltaW != 0, but bounded and logged, 50-1000 steps by contract) spent
+*only* on the exact tie jacob_cov cannot break. Per the Zero-Training
+Contract's own requirement ("must always be possible to answer how much
+PROBE adds over PURE alone, for what additional cost"), this reports that
+cost -- extra training steps spent per decision -- next to whatever
+he-recall / regret improvement it buys, on the same locked TEST split
+used throughout this project. The comparison is against the *raw*
+zc_jacobcov heuristic (already known to be 0/10 on tasks where "he" is
+truly best) and the two failed static tie-breaks, not against FULL
+TRAINING -- that would defeat the point of a cheap PURE-mode heuristic.
+"""
+from __future__ import annotations
+
+import sys
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
+
+import numpy as np
+
+from precog.experiment_db import load_dataframe, record_gate_evaluation
+from precog.meta_predictor import (
+ ProbeTieBreakPredictor,
+ TieBreakHeuristicPredictor,
+ ZeroCostHeuristicPredictor,
+ compute_candidate_zero_cost,
+)
+from precog.model import InitMethod, architecture_from_row
+from precog.modes import TrainingConfig
+from precog.reporting import export_csv_snapshots, write_report
+from precog.taskgen import generate, task_config_from_row
+
+PROBE_STEPS = 50 # the cheapest PROBE-mode call the Zero-Training Contract allows (docs.md §5)
+NON_CONVERGENCE_PENALTY = 800 * 2
+
+
+def _prepare(df):
+ df = df.copy()
+ df["steps_to_threshold"] = df["steps_to_threshold"].fillna(NON_CONVERGENCE_PENALTY)
+ return df
+
+
+def _training_by_candidate(group) -> dict[InitMethod, TrainingConfig]:
+ return {
+ InitMethod(row["training.init_method"]): TrainingConfig(
+ learning_rate=float(row["training.learning_rate"]),
+ batch_size=int(row["training.batch_size"]),
+ optimizer=row["training.optimizer"],
+ weight_decay=float(row["training.weight_decay"]),
+ init_method=InitMethod(row["training.init_method"]),
+ )
+ for _, row in group.iterrows()
+ }
+
+
+def _evaluate(name: str, predictor, test_df, needs_probe_context: bool = False) -> dict:
+ hits = 0
+ regrets = []
+ probe_costs = []
+ he_true_hits = he_true_total = 0
+ rows = []
+ n = 0
+ for seed, group in test_df.groupby("seed"):
+ n += 1
+ features_row = group.iloc[[0]]
+ best_row = group.loc[group["steps_to_threshold"].idxmin()]
+ true_best_init = best_row["training.init_method"]
+ true_best_steps = best_row["steps_to_threshold"]
+
+ task_config = task_config_from_row(features_row.iloc[0])
+ architecture = architecture_from_row(features_row.iloc[0])
+ x, y, _ = generate(task_config)
+ zc_by_candidate = compute_candidate_zero_cost(architecture, task_config.input_dim, x, y)
+
+ if needs_probe_context:
+ rec = predictor.recommend(
+ features_row, zc_by_candidate,
+ architecture=architecture, x=x, y=y,
+ training_by_candidate=_training_by_candidate(group),
+ )
+ else:
+ rec = predictor.recommend(features_row, zc_by_candidate)
+ probe_costs.append(rec.probe_cost_steps)
+
+ hit = rec.recommended_init.value == true_best_init
+ hits += int(hit)
+ if true_best_init == "he":
+ he_true_total += 1
+ he_true_hits += int(hit)
+
+ predicted_steps = group.loc[
+ group["training.init_method"] == rec.recommended_init.value, "steps_to_threshold"
+ ].iloc[0]
+ regret = predicted_steps - true_best_steps
+ regrets.append(regret)
+ rows.append({
+ "seed": seed, "true_best_init": true_best_init, "predicted_init": rec.recommended_init.value,
+ "true_best_steps": true_best_steps, "predicted_steps": predicted_steps,
+ "regret": regret, "probe_cost": rec.probe_cost_steps,
+ })
+
+ accuracy = hits / n
+ mean_regret = float(np.mean(regrets))
+ mean_probe_cost = float(np.mean(probe_costs))
+ he_recall = he_true_hits / he_true_total if he_true_total else float("nan")
+ print(f"{name:<26} accuracy={accuracy:.0%} ({hits}/{n}) "
+ f"he_recall={he_recall:.0%} ({he_true_hits}/{he_true_total}) "
+ f"mean_regret={mean_regret:+.1f} steps mean_probe_cost={mean_probe_cost:.1f} steps")
+ return {
+ "name": name, "accuracy": accuracy, "hits": hits, "n": n,
+ "mean_regret": mean_regret, "mean_probe_cost": mean_probe_cost,
+ "he_recall": he_recall, "he_true_hits": he_true_hits, "he_true_total": he_true_total,
+ "rows": rows,
+ }
+
+
+def main() -> None:
+ test_df = _prepare(load_dataframe(split="test"))
+ print(f"test (locked): {len(test_df)} rows ({test_df['seed'].nunique()} tasks)\n")
+
+ raw = ZeroCostHeuristicPredictor("jacob_cov", higher_is_better=False)
+ tiebreak_raw = TieBreakHeuristicPredictor(primary_proxy="jacob_cov", secondary_proxy="gradient_norm")
+ probe = ProbeTieBreakPredictor(primary_proxy="jacob_cov", probe_steps=PROBE_STEPS)
+
+ r_raw = _evaluate("zc_jacobcov", raw, test_df)
+ r_tiebreak = _evaluate("zc_jacobcov_tiebreak", tiebreak_raw, test_df)
+ r_probe = _evaluate("zc_jacobcov_probetiebreak", probe, test_df, needs_probe_context=True)
+ results = [r_raw, r_tiebreak, r_probe]
+
+ full_training_mean_steps = float(test_df.groupby("seed")["steps_to_threshold"].mean().mean())
+ probe_overhead_pct = 100 * r_probe["mean_probe_cost"] / full_training_mean_steps
+ # he-recall and regret are asked separately: this project's own
+ # methodology (scripts/compare_meta_predictors.py) picks winners by
+ # regret first because a method that's "more often right on a narrow
+ # sub-case" can still be a net-worse decision rule overall -- exactly
+ # what a naive "he-recall went up" reading would miss here.
+ raises_he_recall = r_probe["he_true_hits"] > r_raw["he_true_hits"]
+ beats_on_regret = r_probe["mean_regret"] < r_raw["mean_regret"]
+ net_win = raises_he_recall and beats_on_regret
+
+ record_gate_evaluation(
+ generation="v1-probe-tiebreak", gate_number=1, metric_name="he_recall_probe_tiebreak",
+ metric_value=r_probe["he_recall"] if not np.isnan(r_probe["he_recall"]) else 0.0,
+ threshold=r_raw["he_recall"] if not np.isnan(r_raw["he_recall"]) else 0.0,
+ n_samples=r_probe["he_true_total"],
+ notes=f"raw_he_hits={r_raw['he_true_hits']}/{r_raw['he_true_total']}, "
+ f"tiebreak_he_hits={r_tiebreak['he_true_hits']}/{r_tiebreak['he_true_total']}, "
+ f"probe_he_hits={r_probe['he_true_hits']}/{r_probe['he_true_total']}, "
+ f"probe_steps={PROBE_STEPS}, mean_probe_cost_steps={r_probe['mean_probe_cost']:.1f}, "
+ f"overhead_pct_of_mean_full_training={probe_overhead_pct:.1f}",
+ )
+
+ results_table = "\n".join(
+ f"| {r['name']} | {r['accuracy']:.0%} ({r['hits']}/{r['n']}) | "
+ f"{r['he_recall']:.0%} ({r['he_true_hits']}/{r['he_true_total']}) | "
+ f"{r['mean_regret']:+.1f} | {r['mean_probe_cost']:.1f} |"
+ for r in results
+ )
+ probe_detail = "\n".join(
+ f"| {r['seed']} | {r['true_best_init']} | {r['predicted_init']} | {r['regret']:.0f} | {r['probe_cost']} |"
+ for r in r_probe["rows"]
+ )
+ worst = max(r_probe["rows"], key=lambda r: r["regret"])
+ report = f"""## Method
+
+Three candidates evaluated once each on the identical locked TEST split
+({len(test_df)} rows, {test_df['seed'].nunique()} tasks) used throughout this
+project:
+
+- `zc_jacobcov` -- the raw heuristic, already known to never recommend "he"
+ (0/{r_raw['he_true_total']} on tasks where "he" is truly best).
+- `zc_jacobcov_tiebreak` -- the first static-secondary-proxy fix attempt
+ (gradient_norm), from scripts/compare_meta_predictors.py.
+- `zc_jacobcov_probetiebreak` -- this run's new candidate
+ (precog.meta_predictor.ProbeTieBreakPredictor): on the exact tie
+ jacob_cov cannot break, spends a real, bounded PROBE-mode budget
+ ({PROBE_STEPS} steps per tied candidate, docs.md §5) and picks whichever
+ ends with the lower loss, instead of another PURE-mode secondary proxy.
+
+Per the Zero-Training Contract (docs.md §5: "must always be possible to
+answer how much PROBE adds over PURE alone, for what additional cost"),
+the table below reports that cost -- mean extra training steps spent per
+decision -- next to accuracy, he-recall (of the {r_raw['he_true_total']}
+test tasks where "he" is genuinely the fastest choice) and regret.
+
+## Results
+
+| candidate | accuracy | he-recall | mean regret (steps) | mean probe cost (steps) |
+|---|---:|---:|---:|---:|
+{results_table}
+
+PROBE overhead: {r_probe['mean_probe_cost']:.0f} steps/decision on average,
+against a mean {full_training_mean_steps:.0f}-step FULL TRAINING run on this
+split ({probe_overhead_pct:.1f}% of it). Since jacob_cov ties on every
+single test task (the blind spot is structural, not occasional), this is
+also the candidate's *total* added cost -- there is no untied case to
+amortize it against.
+
+| seed | true best init | predicted (probetiebreak) | regret (steps) | probe cost (steps) |
+|---|---|---|---:|---:|
+{probe_detail}
+
+## Verdict
+
+He-recall {"rises" if raises_he_recall else "does not improve"} from
+{r_raw['he_recall']:.0%} ({r_raw['he_true_hits']}/{r_raw['he_true_total']}) to
+{r_probe['he_recall']:.0%} ({r_probe['he_true_hits']}/{r_probe['he_true_total']}) with the PROBE
+tie-break -- unlike the two static secondary-proxy attempts, spending real
+(bounded) training budget on the exact tie can see the Xavier/He difference
+at all, because it is the only one of the three that isn't a
+positive-rescaling-invariant statistic by construction.
+
+But regret -- this project's primary metric, precisely because it catches
+what accuracy alone can't (scripts/compare_meta_predictors.py) -- gets
+**worse**, not better: {r_raw['mean_regret']:+.1f} steps (raw) to
+{r_probe['mean_regret']:+.1f} steps (PROBE tie-break), on top of
+{r_probe['mean_probe_cost']:.0f} extra steps/decision spent
+({probe_overhead_pct:.1f}% of a mean FULL TRAINING run) to get there.
+{PROBE_STEPS} steps is enough to sometimes make "he" look locally better
+than Xavier, but not enough to foresee the cases named in this project's
+own README §1 where an init that looks fine early on never actually
+converges within budget: the single worst case here is seed {worst['seed']}
+(true best is {worst['true_best_init']} at {worst['true_best_steps']:.0f} steps, the
+probe picks "{worst['predicted_init']}", which needs {worst['predicted_steps']:.0f} steps to
+reach threshold -- regret {worst['regret']:+.0f}). Net: **{"a genuine improvement" if net_win else "NOT a net improvement"}**
+-- {"he-recall and regret both move in the right direction." if net_win else "it trades a narrow, cosmetic fix (jacob_cov's 'never recommends he' symptom) for a worse net decision rule, the same shape of finding as this project's other ideas that looked reasonable but underperformed a simpler baseline (LSUV init, active sampling) -- see source.md pillar 4 and §5 above."}
+"""
+ export_csv_snapshots()
+ report_path = write_report(
+ "explore_probe_tiebreak", "PROBE-Mode Tie-Break for the Xavier/He Blind Spot", report
+ )
+ print(f"\nReport written to {report_path.relative_to(report_path.parents[2])}")
+
+
+if __name__ == "__main__":
+ main()