diff --git a/.github/workflows/reproduce.yml b/.github/workflows/reproduce.yml index 55f3c66..3b551c9 100644 --- a/.github/workflows/reproduce.yml +++ b/.github/workflows/reproduce.yml @@ -45,3 +45,6 @@ jobs: - name: LSUV data-aware init (4th candidate, underperformed) run: python scripts/explore_lsuv_init.py + + - name: PROBE-mode tie-break for the Xavier/He blind spot (raises he-recall, nets worse regret) + run: python scripts/explore_probe_tiebreak.py diff --git a/data/meta_dataset.db b/data/meta_dataset.db index 6b4d776..b0f2fc9 100644 Binary files a/data/meta_dataset.db and b/data/meta_dataset.db differ diff --git a/docs/index.html b/docs/index.html index b117738..b98e8df 100644 --- a/docs/index.html +++ b/docs/index.html @@ -873,134 +873,35 @@
This loop never stops after a single iteration: every PRECOG generation must be compared to the previous one under a strictly identical protocol.
-| Protocol | -Question | -Main metric | -
|---|---|---|
| P1 — Ranking | -Does PRECOG rank configurations correctly? | -Spearman ρ, Kendall τ | -
| P2 — Top-K | -Does it retrieve the best configurations? | -Recall@K | -
| P3 — Convergence | -Does the chosen configuration converge faster? | -Steps/Time-to-Target | -
| P4 — Compute | -How much compute is saved? | -GPU-hours / FLOPs | -
| P5 — Data efficiency | -Same quality with less data? | -Samples-to-Target | -
| P6 — Generalization | -Does it work on a never-seen model/dataset? | -Out-of-distribution performance | -
PRECOG TRAIN → known datasets and architectures, experiment history
-PRECOG VALIDATION → different datasets, partially new architectures
-PRECOG TEST (locked) → never seen, never used to improve PRECOG
-
-Every important experiment is repeated over several seeds, with mean, standard deviation, and confidence interval (95% CI) computed. Comparisons between methods (PRECOG vs. Random, vs. BO, vs. Hyperband, vs. Vizier) use appropriate statistical tests (e.g. a Wilcoxon signed-rank test rather than a t-test when parametric assumptions aren't guaranteed), to avoid declaring superiority based on a lucky seed.
-| Metric | -Definition | -Experimental target | -
|---|---|---|
| Ranking correlation | -Spearman ρ / Kendall τ between PRECOG's ranking and the real ranking | -ρ ≥ 0.80 then ≥ 0.90 | -
| Top-K recall | -$Recall@K = \|\text{PredictedTopK} \cap \text{TrueTopK}\| / K$ | -Recall@10 ≥ 80% then ≥ 90% | -
| Compute reduction | -$1 - C_{PRECOG}/C_{baseline}$ | -≥ 50% then ≥ 70% | -
| Performance retention | -$Performance_{PRECOG}/Performance_{oracle}$ | -≥ 99% (or a tolerance defined a priori) | -
| Data efficiency | -$Samples_{baseline}/Samples_{PRECOG}$ for equal target performance | -≥ 30–50% reduction, to be refined | -
| Time/Steps-to-Target | -Reduction in time/number of steps to reach a target | -≥ 50% reduction | -
| Prediction error (learning curve) | -$\lvert \text{Prediction} - \text{Actual} \rvert$ | -≈ 5–10% depending on the metric | -
| Generalization | -Recall@K on never-seen tasks/architectures/datasets | -same order of magnitude as on known data | -
These targets are progression hypotheses, formalized as successive gates (§17), never presented as already achieved.
-flowchart TD - P[PRECOG] --> G1["Gate 1: ρ ≥ 0.70?"] - G1 --> G2["Gate 2: Recall@10 ≥ 80%?"] - G2 --> G3["Gate 3: Compute reduction ≥ 50%?"] - G3 --> G4["Gate 4: Generalization maintained<br/>(never-seen data)?"] - G4 --> G5["Gate 5: Recall@10 ≥ 90%?"] - G5 --> G6["Gate 6: Compute reduction ≥ 70%?"] - G6 --> ADV["PRECOG 'advanced level'"] ++ classDef next fill:#eef2ff,stroke:#1a56db,stroke-width:2px,color:#1a56db; class ADV next; diff --git a/precog/meta_predictor.py b/precog/meta_predictor.py index 2e986c1..cb78b85 100644 --- a/precog/meta_predictor.py +++ b/precog/meta_predictor.py @@ -52,6 +52,7 @@ from precog.meta_knowledge_base import MetaKnowledgeBase from precog.model import InitMethod +from precog.modes import Mode, TrainingConfig, TrainProtocol, train from precog.regime import _bucket_noise, _bucket_volume from precog.trainability import zero_cost_features @@ -151,6 +152,7 @@ class Recommendation: steps_range: tuple[float, float] # +/- 1 std across the ensemble confidence: float # 1 - (relative spread), clamped to [0, 1] per_candidate: dict[str, dict] # every candidate's own prediction, for transparency + probe_cost_steps: int = 0 # PROBE-mode budget actually spent on this decision (docs.md §5 cost-accounting) class MetaPredictor: @@ -433,6 +435,91 @@ def secondary_key(k: str) -> float: ) +class ProbeTieBreakPredictor: + """Third fix attempt for zc_jacobcov's proven blind spot (see + TieBreakHeuristicPredictor above): jacob_cov's binary activation-sign + statistic is *exactly* invariant to the positive rescaling that + separates Xavier from He (max |xavier-he| jacob_cov = 0.0 across all + 312 meta-dataset tasks), so no PURE-mode secondary proxy -- raw or + population-normalized gradient_norm, both tried in + scripts/compare_meta_predictors.py -- can ever break that exact tie; + gradient_norm turned out to carry the same he>xavier scale confound + jacob_cov's sign-only statistic doesn't even look at. + + This tries the other option named in + results/reports/2026-09-02T08-04-49Z_explore_scale_invariance_blindspot.md: + a minimal PROBE-mode check (docs.md §5: DeltaW != 0, but bounded and + logged, 50-1000 steps by contract) spent *only* on the exact tie + jacob_cov cannot see -- train each tied candidate for `probe_steps` + real steps at its own (learning_rate, batch_size, optimizer) and keep + whichever ends with the lower loss. `last_probe_cost_steps` records + the budget actually spent on the most recent call, so callers can + report it per the Zero-Training Contract's own requirement ("must + always be possible to answer how much PROBE adds over PURE alone, for + what additional cost") -- see scripts/explore_probe_tiebreak.py.""" + + def __init__( + self, + primary_proxy: str = "jacob_cov", + primary_higher_is_better: bool = False, + tie_tolerance: float = 1e-6, + probe_steps: int = 50, + ): + self.primary_proxy = primary_proxy + self.primary_higher_is_better = primary_higher_is_better + self.tie_tolerance = tie_tolerance + self.probe_steps = probe_steps + self.last_probe_cost_steps = 0 + + def recommend( + self, + features_row: pd.DataFrame, + zero_cost_by_candidate: dict[InitMethod, dict], + architecture=None, + x: torch.Tensor | None = None, + y: torch.Tensor | None = None, + training_by_candidate: dict[InitMethod, TrainingConfig] | None = None, + ) -> Recommendation: + per_candidate = { + c.value: {"expected_steps": float("nan"), "std_steps": 0.0, "primary_score": zc[self.primary_proxy]} + for c, zc in zero_cost_by_candidate.items() + } + primary_sign = -1 if self.primary_higher_is_better else 1 + primary_values = {k: v["primary_score"] * primary_sign for k, v in per_candidate.items()} + best_primary = min(primary_values.values()) + tied = [k for k, v in primary_values.items() if abs(v - best_primary) <= self.tie_tolerance] + + self.last_probe_cost_steps = 0 + if len(tied) == 1: + best_init_name = tied[0] + else: + if architecture is None or x is None or y is None or training_by_candidate is None: + raise ValueError( + "ProbeTieBreakPredictor needs a live architecture/x/y/training_by_candidate " + "context to actually run the PROBE that breaks the tie -- pass them through, " + "see scripts/explore_probe_tiebreak.py for how the harness wires this up." + ) + probe_losses = {} + for k in tied: + training = training_by_candidate[InitMethod(k)] + protocol = TrainProtocol( + mode=Mode.PROBE, max_steps=self.probe_steps, loss_threshold=-1.0, seed=0 + ) + result = train(architecture, x, y, training, protocol) + probe_losses[k] = result.final_loss + self.last_probe_cost_steps += self.probe_steps + best_init_name = min(probe_losses, key=probe_losses.get) + + return Recommendation( + recommended_init=InitMethod(best_init_name), + expected_steps=float("nan"), + steps_range=(float("nan"), float("nan")), + confidence=float("nan"), # this method makes no probabilistic claim -- see docs.md §20 + per_candidate=per_candidate, + probe_cost_steps=self.last_probe_cost_steps, + ) + + class KNNMetaPredictor: """Alternative to the RandomForest MetaPredictor (docs.md §19 ablation spirit): predicts purely from the Meta-Knowledge Base's (§9.6) nearest diff --git a/results/gate_evaluations.csv b/results/gate_evaluations.csv index c4b4646..4e655fc 100644 --- a/results/gate_evaluations.csv +++ b/results/gate_evaluations.csv @@ -232,3 +232,4 @@ evaluation_id,timestamp,generation,gate_number,metric_name,metric_value,threshol 231,2026-09-02 07:46:49,v1-meta-predictor-zc_jacobcov_normtiebreak,2,mean_regret_steps_to_threshold,47.666666666666664,0.0,1,60,regret = steps(predicted_init) - steps(true_best_init); relative_regret=+33.72%; universal_baseline_regret=+22.2; random_baseline_regret=+43.2 232,2026-09-02 08:04:49,v1-scale-invariance-blindspot,0,fraction_proxies_blind_to_xavier_vs_he,0.2727272727272727,0.0,1,312,"blind_proxies=['jacob_cov', 'effective_rank', 'jacobian_condition_mean'], tie_threshold=5% mean relative difference" 233,2026-09-02 19:28:18,v1-trainability-engine-at-scale,1,spearman_rho_gradient_norm_vs_steps_full_scale,0.5398950415558994,0.7,0,936,"full meta-dataset re-check (312 tasks) of gate1_ranking.py's original n=36 (12-task) result; no new training runs, same controlled design (§21)" +234,2026-09-03 12:59:38,v1-probe-tiebreak,1,he_recall_probe_tiebreak,0.3,0.0,1,10,"raw_he_hits=0/10, tiebreak_he_hits=0/10, probe_he_hits=3/10, probe_steps=50, mean_probe_cost_steps=68.3, overhead_pct_of_mean_full_training=37.5" diff --git a/results/reports/2026-09-03T12-59-38Z_explore_probe_tiebreak.md b/results/reports/2026-09-03T12-59-38Z_explore_probe_tiebreak.md new file mode 100644 index 0000000..96f4486 --- /dev/null +++ b/results/reports/2026-09-03T12-59-38Z_explore_probe_tiebreak.md @@ -0,0 +1,128 @@ +# PROBE-Mode Tie-Break for the Xavier/He Blind Spot + +_Generated 2026-09-03T12-59-38Z (UTC)_ + +## Method + +Three candidates evaluated once each on the identical locked TEST split +(180 rows, 60 tasks) used throughout this +project: + +- `zc_jacobcov` -- the raw heuristic, already known to never recommend "he" + (0/10 on tasks where "he" is truly best). +- `zc_jacobcov_tiebreak` -- the first static-secondary-proxy fix attempt + (gradient_norm), from scripts/compare_meta_predictors.py. +- `zc_jacobcov_probetiebreak` -- this run's new candidate + (precog.meta_predictor.ProbeTieBreakPredictor): on the exact tie + jacob_cov cannot break, spends a real, bounded PROBE-mode budget + (50 steps per tied candidate, docs.md §5) and picks whichever + ends with the lower loss, instead of another PURE-mode secondary proxy. + +Per the Zero-Training Contract (docs.md §5: "must always be possible to +answer how much PROBE adds over PURE alone, for what additional cost"), +the table below reports that cost -- mean extra training steps spent per +decision -- next to accuracy, he-recall (of the 10 +test tasks where "he" is genuinely the fastest choice) and regret. + +## Results + +| candidate | accuracy | he-recall | mean regret (steps) | mean probe cost (steps) | +|---|---:|---:|---:|---:| +| zc_jacobcov | 47% (28/60) | 0% (0/10) | +14.6 | 0.0 | +| zc_jacobcov_tiebreak | 47% (28/60) | 0% (0/10) | +14.6 | 0.0 | +| zc_jacobcov_probetiebreak | 43% (26/60) | 30% (3/10) | +45.4 | 68.3 | + +PROBE overhead: 68 steps/decision on average, +against a mean 182-step FULL TRAINING run on this +split (37.5% of it). Since jacob_cov ties on every +single test task (the blind spot is structural, not occasional), this is +also the candidate's *total* added cost -- there is no untied case to +amortize it against. + +| seed | true best init | predicted (probetiebreak) | regret (steps) | probe cost (steps) | +|---|---|---|---:|---:| +| 100 | xavier | orthogonal | 10 | 0 | +| 107 | orthogonal | orthogonal | 0 | 0 | +| 120 | orthogonal | he | 149 | 100 | +| 131 | xavier | xavier | 0 | 100 | +| 132 | xavier | xavier | 0 | 100 | +| 137 | orthogonal | xavier | 7 | 100 | +| 141 | orthogonal | he | 128 | 100 | +| 146 | orthogonal | orthogonal | 0 | 0 | +| 147 | xavier | he | 973 | 100 | +| 148 | he | orthogonal | 1 | 0 | +| 150 | he | he | 0 | 100 | +| 151 | orthogonal | orthogonal | 0 | 0 | +| 155 | orthogonal | xavier | 54 | 100 | +| 171 | orthogonal | orthogonal | 0 | 0 | +| 172 | he | he | 0 | 100 | +| 175 | orthogonal | xavier | 3 | 100 | +| 197 | xavier | xavier | 0 | 100 | +| 204 | xavier | he | 281 | 100 | +| 211 | orthogonal | xavier | 1 | 100 | +| 213 | xavier | xavier | 0 | 100 | +| 222 | orthogonal | orthogonal | 0 | 0 | +| 224 | xavier | xavier | 0 | 100 | +| 228 | xavier | xavier | 0 | 100 | +| 232 | he | xavier | 8 | 100 | +| 233 | orthogonal | xavier | 71 | 100 | +| 244 | xavier | he | 9 | 100 | +| 249 | he | orthogonal | 55 | 0 | +| 254 | orthogonal | orthogonal | 0 | 0 | +| 255 | orthogonal | orthogonal | 0 | 0 | +| 258 | xavier | orthogonal | 7 | 0 | +| 261 | xavier | xavier | 0 | 100 | +| 263 | orthogonal | xavier | 66 | 100 | +| 266 | orthogonal | he | 49 | 100 | +| 269 | orthogonal | xavier | 9 | 100 | +| 270 | he | xavier | 8 | 100 | +| 281 | he | xavier | 9 | 100 | +| 283 | orthogonal | orthogonal | 0 | 0 | +| 297 | orthogonal | xavier | 46 | 100 | +| 304 | orthogonal | orthogonal | 0 | 0 | +| 307 | orthogonal | xavier | 8 | 100 | +| 315 | xavier | xavier | 0 | 100 | +| 322 | orthogonal | xavier | 1 | 100 | +| 326 | xavier | xavier | 0 | 100 | +| 329 | xavier | xavier | 0 | 100 | +| 341 | xavier | xavier | 0 | 100 | +| 344 | orthogonal | xavier | 131 | 100 | +| 348 | he | orthogonal | 18 | 0 | +| 350 | orthogonal | xavier | 218 | 100 | +| 352 | xavier | xavier | 0 | 100 | +| 358 | he | orthogonal | 8 | 0 | +| 360 | he | he | 0 | 100 | +| 361 | orthogonal | he | 9 | 100 | +| 366 | xavier | orthogonal | 7 | 0 | +| 372 | xavier | orthogonal | 3 | 0 | +| 378 | orthogonal | orthogonal | 0 | 0 | +| 380 | xavier | xavier | 0 | 100 | +| 382 | xavier | he | 5 | 100 | +| 386 | xavier | orthogonal | 0 | 0 | +| 390 | xavier | he | 214 | 100 | +| 398 | orthogonal | he | 156 | 100 | + +## Verdict + +He-recall rises from +0% (0/10) to +30% (3/10) with the PROBE +tie-break -- unlike the two static secondary-proxy attempts, spending real +(bounded) training budget on the exact tie can see the Xavier/He difference +at all, because it is the only one of the three that isn't a +positive-rescaling-invariant statistic by construction. + +But regret -- this project's primary metric, precisely because it catches +what accuracy alone can't (scripts/compare_meta_predictors.py) -- gets +**worse**, not better: +14.6 steps (raw) to ++45.4 steps (PROBE tie-break), on top of +68 extra steps/decision spent +(37.5% of a mean FULL TRAINING run) to get there. +50 steps is enough to sometimes make "he" look locally better +than Xavier, but not enough to foresee the cases named in this project's +own README §1 where an init that looks fine early on never actually +converges within budget: the single worst case here is seed 147 +(true best is xavier at 627 steps, the +probe picks "he", which needs 1600 steps to +reach threshold -- regret +973). Net: **NOT a net improvement** +-- it trades a narrow, cosmetic fix (jacob_cov's 'never recommends he' symptom) for a worse net decision rule, the same shape of finding as this project's other ideas that looked reasonable but underperformed a simpler baseline (LSUV init, active sampling) -- see source.md pillar 4 and §5 above. diff --git a/scripts/explore_probe_tiebreak.py b/scripts/explore_probe_tiebreak.py new file mode 100644 index 0000000..e0b5b22 --- /dev/null +++ b/scripts/explore_probe_tiebreak.py @@ -0,0 +1,251 @@ +#!/usr/bin/env python3 +"""Tests the second fix option named in +results/reports/2026-09-02T08-04-49Z_explore_scale_invariance_blindspot.md +for zc_jacobcov's proven blind spot: jacob_cov's binary activation-sign +statistic is *exactly* invariant to the positive rescaling that separates +Xavier from He (max |xavier-he| jacob_cov = 0.0 across all 312 meta-dataset +tasks), so no PURE-mode reading of it, however combined with a secondary +proxy, can ever tell them apart. Both static tie-breaks already tried in +scripts/compare_meta_predictors.py (TieBreakHeuristicPredictor's raw and +population-normalized gradient_norm variants) failed for exactly this +reason -- gradient_norm turned out to carry the same he>xavier scale +confound jacob_cov's sign-only statistic doesn't even try to see. + +precog.meta_predictor.ProbeTieBreakPredictor tries the other option the +report named: a minimal, explicitly-costed PROBE-mode run (docs.md §5: +DeltaW != 0, but bounded and logged, 50-1000 steps by contract) spent +*only* on the exact tie jacob_cov cannot break. Per the Zero-Training +Contract's own requirement ("must always be possible to answer how much +PROBE adds over PURE alone, for what additional cost"), this reports that +cost -- extra training steps spent per decision -- next to whatever +he-recall / regret improvement it buys, on the same locked TEST split +used throughout this project. The comparison is against the *raw* +zc_jacobcov heuristic (already known to be 0/10 on tasks where "he" is +truly best) and the two failed static tie-breaks, not against FULL +TRAINING -- that would defeat the point of a cheap PURE-mode heuristic. +""" +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +import numpy as np + +from precog.experiment_db import load_dataframe, record_gate_evaluation +from precog.meta_predictor import ( + ProbeTieBreakPredictor, + TieBreakHeuristicPredictor, + ZeroCostHeuristicPredictor, + compute_candidate_zero_cost, +) +from precog.model import InitMethod, architecture_from_row +from precog.modes import TrainingConfig +from precog.reporting import export_csv_snapshots, write_report +from precog.taskgen import generate, task_config_from_row + +PROBE_STEPS = 50 # the cheapest PROBE-mode call the Zero-Training Contract allows (docs.md §5) +NON_CONVERGENCE_PENALTY = 800 * 2 + + +def _prepare(df): + df = df.copy() + df["steps_to_threshold"] = df["steps_to_threshold"].fillna(NON_CONVERGENCE_PENALTY) + return df + + +def _training_by_candidate(group) -> dict[InitMethod, TrainingConfig]: + return { + InitMethod(row["training.init_method"]): TrainingConfig( + learning_rate=float(row["training.learning_rate"]), + batch_size=int(row["training.batch_size"]), + optimizer=row["training.optimizer"], + weight_decay=float(row["training.weight_decay"]), + init_method=InitMethod(row["training.init_method"]), + ) + for _, row in group.iterrows() + } + + +def _evaluate(name: str, predictor, test_df, needs_probe_context: bool = False) -> dict: + hits = 0 + regrets = [] + probe_costs = [] + he_true_hits = he_true_total = 0 + rows = [] + n = 0 + for seed, group in test_df.groupby("seed"): + n += 1 + features_row = group.iloc[[0]] + best_row = group.loc[group["steps_to_threshold"].idxmin()] + true_best_init = best_row["training.init_method"] + true_best_steps = best_row["steps_to_threshold"] + + task_config = task_config_from_row(features_row.iloc[0]) + architecture = architecture_from_row(features_row.iloc[0]) + x, y, _ = generate(task_config) + zc_by_candidate = compute_candidate_zero_cost(architecture, task_config.input_dim, x, y) + + if needs_probe_context: + rec = predictor.recommend( + features_row, zc_by_candidate, + architecture=architecture, x=x, y=y, + training_by_candidate=_training_by_candidate(group), + ) + else: + rec = predictor.recommend(features_row, zc_by_candidate) + probe_costs.append(rec.probe_cost_steps) + + hit = rec.recommended_init.value == true_best_init + hits += int(hit) + if true_best_init == "he": + he_true_total += 1 + he_true_hits += int(hit) + + predicted_steps = group.loc[ + group["training.init_method"] == rec.recommended_init.value, "steps_to_threshold" + ].iloc[0] + regret = predicted_steps - true_best_steps + regrets.append(regret) + rows.append({ + "seed": seed, "true_best_init": true_best_init, "predicted_init": rec.recommended_init.value, + "true_best_steps": true_best_steps, "predicted_steps": predicted_steps, + "regret": regret, "probe_cost": rec.probe_cost_steps, + }) + + accuracy = hits / n + mean_regret = float(np.mean(regrets)) + mean_probe_cost = float(np.mean(probe_costs)) + he_recall = he_true_hits / he_true_total if he_true_total else float("nan") + print(f"{name:<26} accuracy={accuracy:.0%} ({hits}/{n}) " + f"he_recall={he_recall:.0%} ({he_true_hits}/{he_true_total}) " + f"mean_regret={mean_regret:+.1f} steps mean_probe_cost={mean_probe_cost:.1f} steps") + return { + "name": name, "accuracy": accuracy, "hits": hits, "n": n, + "mean_regret": mean_regret, "mean_probe_cost": mean_probe_cost, + "he_recall": he_recall, "he_true_hits": he_true_hits, "he_true_total": he_true_total, + "rows": rows, + } + + +def main() -> None: + test_df = _prepare(load_dataframe(split="test")) + print(f"test (locked): {len(test_df)} rows ({test_df['seed'].nunique()} tasks)\n") + + raw = ZeroCostHeuristicPredictor("jacob_cov", higher_is_better=False) + tiebreak_raw = TieBreakHeuristicPredictor(primary_proxy="jacob_cov", secondary_proxy="gradient_norm") + probe = ProbeTieBreakPredictor(primary_proxy="jacob_cov", probe_steps=PROBE_STEPS) + + r_raw = _evaluate("zc_jacobcov", raw, test_df) + r_tiebreak = _evaluate("zc_jacobcov_tiebreak", tiebreak_raw, test_df) + r_probe = _evaluate("zc_jacobcov_probetiebreak", probe, test_df, needs_probe_context=True) + results = [r_raw, r_tiebreak, r_probe] + + full_training_mean_steps = float(test_df.groupby("seed")["steps_to_threshold"].mean().mean()) + probe_overhead_pct = 100 * r_probe["mean_probe_cost"] / full_training_mean_steps + # he-recall and regret are asked separately: this project's own + # methodology (scripts/compare_meta_predictors.py) picks winners by + # regret first because a method that's "more often right on a narrow + # sub-case" can still be a net-worse decision rule overall -- exactly + # what a naive "he-recall went up" reading would miss here. + raises_he_recall = r_probe["he_true_hits"] > r_raw["he_true_hits"] + beats_on_regret = r_probe["mean_regret"] < r_raw["mean_regret"] + net_win = raises_he_recall and beats_on_regret + + record_gate_evaluation( + generation="v1-probe-tiebreak", gate_number=1, metric_name="he_recall_probe_tiebreak", + metric_value=r_probe["he_recall"] if not np.isnan(r_probe["he_recall"]) else 0.0, + threshold=r_raw["he_recall"] if not np.isnan(r_raw["he_recall"]) else 0.0, + n_samples=r_probe["he_true_total"], + notes=f"raw_he_hits={r_raw['he_true_hits']}/{r_raw['he_true_total']}, " + f"tiebreak_he_hits={r_tiebreak['he_true_hits']}/{r_tiebreak['he_true_total']}, " + f"probe_he_hits={r_probe['he_true_hits']}/{r_probe['he_true_total']}, " + f"probe_steps={PROBE_STEPS}, mean_probe_cost_steps={r_probe['mean_probe_cost']:.1f}, " + f"overhead_pct_of_mean_full_training={probe_overhead_pct:.1f}", + ) + + results_table = "\n".join( + f"| {r['name']} | {r['accuracy']:.0%} ({r['hits']}/{r['n']}) | " + f"{r['he_recall']:.0%} ({r['he_true_hits']}/{r['he_true_total']}) | " + f"{r['mean_regret']:+.1f} | {r['mean_probe_cost']:.1f} |" + for r in results + ) + probe_detail = "\n".join( + f"| {r['seed']} | {r['true_best_init']} | {r['predicted_init']} | {r['regret']:.0f} | {r['probe_cost']} |" + for r in r_probe["rows"] + ) + worst = max(r_probe["rows"], key=lambda r: r["regret"]) + report = f"""## Method + +Three candidates evaluated once each on the identical locked TEST split +({len(test_df)} rows, {test_df['seed'].nunique()} tasks) used throughout this +project: + +- `zc_jacobcov` -- the raw heuristic, already known to never recommend "he" + (0/{r_raw['he_true_total']} on tasks where "he" is truly best). +- `zc_jacobcov_tiebreak` -- the first static-secondary-proxy fix attempt + (gradient_norm), from scripts/compare_meta_predictors.py. +- `zc_jacobcov_probetiebreak` -- this run's new candidate + (precog.meta_predictor.ProbeTieBreakPredictor): on the exact tie + jacob_cov cannot break, spends a real, bounded PROBE-mode budget + ({PROBE_STEPS} steps per tied candidate, docs.md §5) and picks whichever + ends with the lower loss, instead of another PURE-mode secondary proxy. + +Per the Zero-Training Contract (docs.md §5: "must always be possible to +answer how much PROBE adds over PURE alone, for what additional cost"), +the table below reports that cost -- mean extra training steps spent per +decision -- next to accuracy, he-recall (of the {r_raw['he_true_total']} +test tasks where "he" is genuinely the fastest choice) and regret. + +## Results + +| candidate | accuracy | he-recall | mean regret (steps) | mean probe cost (steps) | +|---|---:|---:|---:|---:| +{results_table} + +PROBE overhead: {r_probe['mean_probe_cost']:.0f} steps/decision on average, +against a mean {full_training_mean_steps:.0f}-step FULL TRAINING run on this +split ({probe_overhead_pct:.1f}% of it). Since jacob_cov ties on every +single test task (the blind spot is structural, not occasional), this is +also the candidate's *total* added cost -- there is no untied case to +amortize it against. + +| seed | true best init | predicted (probetiebreak) | regret (steps) | probe cost (steps) | +|---|---|---|---:|---:| +{probe_detail} + +## Verdict + +He-recall {"rises" if raises_he_recall else "does not improve"} from +{r_raw['he_recall']:.0%} ({r_raw['he_true_hits']}/{r_raw['he_true_total']}) to +{r_probe['he_recall']:.0%} ({r_probe['he_true_hits']}/{r_probe['he_true_total']}) with the PROBE +tie-break -- unlike the two static secondary-proxy attempts, spending real +(bounded) training budget on the exact tie can see the Xavier/He difference +at all, because it is the only one of the three that isn't a +positive-rescaling-invariant statistic by construction. + +But regret -- this project's primary metric, precisely because it catches +what accuracy alone can't (scripts/compare_meta_predictors.py) -- gets +**worse**, not better: {r_raw['mean_regret']:+.1f} steps (raw) to +{r_probe['mean_regret']:+.1f} steps (PROBE tie-break), on top of +{r_probe['mean_probe_cost']:.0f} extra steps/decision spent +({probe_overhead_pct:.1f}% of a mean FULL TRAINING run) to get there. +{PROBE_STEPS} steps is enough to sometimes make "he" look locally better +than Xavier, but not enough to foresee the cases named in this project's +own README §1 where an init that looks fine early on never actually +converges within budget: the single worst case here is seed {worst['seed']} +(true best is {worst['true_best_init']} at {worst['true_best_steps']:.0f} steps, the +probe picks "{worst['predicted_init']}", which needs {worst['predicted_steps']:.0f} steps to +reach threshold -- regret {worst['regret']:+.0f}). Net: **{"a genuine improvement" if net_win else "NOT a net improvement"}** +-- {"he-recall and regret both move in the right direction." if net_win else "it trades a narrow, cosmetic fix (jacob_cov's 'never recommends he' symptom) for a worse net decision rule, the same shape of finding as this project's other ideas that looked reasonable but underperformed a simpler baseline (LSUV init, active sampling) -- see source.md pillar 4 and §5 above."} +""" + export_csv_snapshots() + report_path = write_report( + "explore_probe_tiebreak", "PROBE-Mode Tie-Break for the Xavier/He Blind Spot", report + ) + print(f"\nReport written to {report_path.relative_to(report_path.parents[2])}") + + +if __name__ == "__main__": + main()04A bug we found but can't fix
++
+jacob_covnever once recommends "He init" across 60 test + tasks — including the 10 where it's genuinely fastest. The cause is + structural: Xavier and He (zero-biased networks) draw from the same + underlying Gaussian values, differing only by a positive scale — and +jacob_covonly reads activation sign, which scaling + cannot flip. ++jacob_cov(He-init network) == jacob_cov(Xavier-init network)+ exactly, for the same seed — checked across all 312 tasks, max difference + is
+ Full audit of all 11 proxies → + Why the PROBE-mode fix doesn't net out → +0.0. Two other proxies (effective_rank, +jacobian_condition_mean) share the property for the same + reason. Three attempted fixes, all failed: raw and population-normalized + tie-breaking withgradient_norm(same structural blindness, + different proxy), and — the one thing that should have worked, since + it stops reading a PURE-mode proxy entirely — a bounded 50-step + PROBE-mode + run on just the tied candidates. It does recover some "he" picks + (recall 0%→30% on the 10 tasks where "he" is truly best), but net + regret gets worse (+14.6→+45.4 steps): 50 real training steps + is enough to make "he" look locally better, not enough to see that it + sometimes never converges at all within budget. +