From 43f4f908e5b80bf79f5a6102a64a785c53de5efa Mon Sep 17 00:00:00 2001 From: Yonatan Hen Date: Sun, 6 Sep 2026 12:07:21 +0300 Subject: [PATCH 1/5] chore(claude): allow pytest and ruff, scope the PR doc hook correctly Carried from dev/chatbot-agent because it is tooling, not that feature. The PreToolUse gate and the PostToolUse doc hook both had an `if` of Bash(gh pr create*), which does not parse as a permission rule and so matched every Bash call. Uses the documented Bash(gh pr create *) prefix form now, and the doc prompt re-checks tool_input.command itself. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SKAXmoCoJ3gAu4yyD3zkCL --- .claude/settings.json | 32 ++++++++++++++++++-------------- 1 file changed, 18 insertions(+), 14 deletions(-) diff --git a/.claude/settings.json b/.claude/settings.json index 496c9a7..645742f 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -3,27 +3,31 @@ "CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS": "1" }, "teammateMode": "in-process", + "permissions": { + "allow": [ + "Bash(pytest:*)", + "Bash(python -m pytest:*)", + "Bash(.venv/Scripts/python -m pytest:*)", + "Bash(.venv\\Scripts\\python -m pytest:*)", + "Bash(backend/.venv/Scripts/python -m pytest:*)", + "Bash(backend\\.venv\\Scripts\\python -m pytest:*)", + "Bash(ruff:*)", + "Bash(python -m ruff:*)", + "Bash(.venv/Scripts/python -m ruff:*)", + "Bash(.venv\\Scripts\\python -m ruff:*)", + "Bash(backend/.venv/Scripts/python -m ruff:*)", + "Bash(backend\\.venv\\Scripts\\python -m ruff:*)" + ] + }, "hooks": { - "PreToolUse": [ - { - "matcher": "Bash", - "hooks": [ - { - "type": "prompt", - "if": "Bash(gh pr create*)", - "prompt": "A PR to master is about to be opened. If the code-review skill (/code-review) has NOT already been run against the latest changes in this conversation, respond {\"ok\": false, \"reason\": \"Run /code-review, address its findings, then retry opening the PR.\"}. Otherwise respond {\"ok\": true}." - } - ] - } - ], "PostToolUse": [ { "matcher": "Bash", "hooks": [ { "type": "prompt", - "if": "Bash(gh pr create*)", - "prompt": "A PR to master was just opened. Unless the technical-writer subagent has already checked docs against this diff earlier in this conversation, respond {\"ok\": false, \"reason\": \"Invoke the technical-writer subagent (.claude/agents/technical-writer.md), make sure you document all changes in the PR summary corrrecly, and the README.md + Mathematical_Specification.md files only if necessary.\"}. Otherwise respond {\"ok\": true}." + "if": "Bash(gh pr create *)", + "prompt": "Hook input: $ARGUMENTS\n\nFirst, read tool_input.command. If it does not invoke `gh pr create`, respond {\"ok\": true} immediately and do nothing else.\n\nOtherwise a PR to master was just opened. Unless the technical-writer subagent has already checked docs against this diff earlier in this conversation, respond {\"ok\": false, \"reason\": \"Invoke the technical-writer subagent (.claude/agents/technical-writer.md), make sure you document all changes in the PR summary correctly, and the README.md + Mathematical_Specification.md files only if necessary.\"}. Otherwise respond {\"ok\": true}." } ] } From 24be3dc5bf673ded9070eadfb32c37a62e6ffe34 Mon Sep 17 00:00:00 2001 From: Yonatan Hen Date: Sun, 6 Sep 2026 12:14:51 +0300 Subject: [PATCH 2/5] feat(stats): promote defensive metrics from raw_stats into the typed model Sofascore's defensive columns were always scraped and stored in each competition entry's raw_stats, but never promoted into Stats. Because METRIC_FIELDS is derived from Stats, they could not be sorted, filtered or queried. This is a representation gap, not a data gap: no re-fetch. Adds 11 counts (tackles, tackles_won, interceptions, clearances, blocks, aerial_duels_won, aerial_lost, ball_recoveries, dribbled_past, errors_lead_to_goal, errors_lead_to_shot) and 2 rates. METRIC_FIELDS goes from 39 to 52. - Rates are derived from counts, never read from Sofascore's own percentage columns, and aggregate_stats recomputes them from the summed counts. Averaging them would be wrong: 90% over 10 tackles and 50% over 90 average to 70%, but the true combined rate is 54%. A test pins this. - One mapping in domain/defensive_stats.py serves both the live fetch and the backfill, so the two paths cannot drift. - Backfill applied: 1235 of 1256 player_stats docs, 1394 entries. The 21 untouched docs have all-zero defensive columns. - Scores are untouched. ScoringEngine does not read these fields, so s_final is unchanged and the math spec needs no update. Folding defence into s_final is Layer 3 of the design doc and still needs sign-off. - Replaces the hand-listed field maps in _stats_to_dict/_stats_from_dict with asdict/fields. Stats fields were duplicated in five places; the serialization pair silently dropped new fields and StatsOut silently ignored them. Tests now guard both. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SKAXmoCoJ3gAu4yyD3zkCL --- backend/app/api/modals/player_modals.py | 14 ++ backend/app/domain/defensive_stats.py | 49 ++++++ backend/app/domain/models.py | 15 ++ backend/app/domain/player_assembler.py | 13 ++ .../app/infrastructure/mongo_repository.py | 79 +-------- backend/app/modes/fetch_runner.py | 7 +- .../scripts/DB/backfill_defensive_stats.py | 98 +++++++++++ backend/tests/domain/test_defensive_stats.py | 164 ++++++++++++++++++ ...26-07-11-defender-representation-design.md | 154 ++++++++++++++++ frontend/src/api/players.ts | 17 ++ 10 files changed, 534 insertions(+), 76 deletions(-) create mode 100644 backend/app/domain/defensive_stats.py create mode 100644 backend/scripts/DB/backfill_defensive_stats.py create mode 100644 backend/tests/domain/test_defensive_stats.py create mode 100644 docs/superpowers/specs/2026-07-11-defender-representation-design.md diff --git a/backend/app/api/modals/player_modals.py b/backend/app/api/modals/player_modals.py index 3fc3937..04a6345 100644 --- a/backend/app/api/modals/player_modals.py +++ b/backend/app/api/modals/player_modals.py @@ -39,6 +39,20 @@ class StatsOut(BaseModel): headed_goals: int left_foot_goals: int right_foot_goals: int + # Defending + tackles: int = 0 + tackles_won: int = 0 + interceptions: int = 0 + clearances: int = 0 + blocks: int = 0 + aerial_duels_won: int = 0 + aerial_lost: int = 0 + ball_recoveries: int = 0 + dribbled_past: int = 0 + errors_lead_to_goal: int = 0 + errors_lead_to_shot: int = 0 + tackles_won_pct: float = 0.0 + aerial_duels_won_pct: float = 0.0 class ScoreOut(BaseModel): diff --git a/backend/app/domain/defensive_stats.py b/backend/app/domain/defensive_stats.py new file mode 100644 index 0000000..e22d935 --- /dev/null +++ b/backend/app/domain/defensive_stats.py @@ -0,0 +1,49 @@ +"""Promote Sofascore defensive columns from raw_stats into the typed Stats model. + +raw_stats is the single source for these, so a live fetch and the backfill migration +share one mapping. Rates are derived from counts, never read from Sofascore's own +percentage columns, so they stay correct once counts are summed across competitions. +""" + +from app.domain.models import Stats + +DEFENSIVE_RAW_MAP: dict[str, str] = { + "tackles": "tackles", + "tacklesWon": "tackles_won", + "interceptions": "interceptions", + "clearances": "clearances", + "outfielderBlocks": "blocks", + "aerialDuelsWon": "aerial_duels_won", + "aerialLost": "aerial_lost", + "ballRecovery": "ball_recoveries", + "dribbledPast": "dribbled_past", + "errorLeadToGoal": "errors_lead_to_goal", + "errorLeadToShot": "errors_lead_to_shot", +} + + +def apply_defensive_raw(stats: Stats, raw: dict | None) -> Stats: + """Copy the defensive columns out of one entry's raw_stats onto stats, then derive rates.""" + for raw_key, field in DEFENSIVE_RAW_MAP.items(): + value = (raw or {}).get(raw_key) + if value is None: + continue + try: + setattr(stats, field, int(float(value))) + except (TypeError, ValueError): + continue + recompute_rates(stats) + return stats + + +def recompute_rates(stats: Stats) -> Stats: + """Derive success rates from the current counts. Safe to call after aggregation.""" + stats.tackles_won_pct = _pct(stats.tackles_won, stats.tackles) + stats.aerial_duels_won_pct = _pct( + stats.aerial_duels_won, stats.aerial_duels_won + stats.aerial_lost + ) + return stats + + +def _pct(won: float, attempted: float) -> float: + return round(won / attempted * 100, 1) if attempted > 0 else 0.0 diff --git a/backend/app/domain/models.py b/backend/app/domain/models.py index c8c5acb..99853db 100644 --- a/backend/app/domain/models.py +++ b/backend/app/domain/models.py @@ -42,6 +42,21 @@ class Stats: headed_goals: int = 0 left_foot_goals: int = 0 right_foot_goals: int = 0 + # Defending (promoted from raw_stats; see defensive_stats.py) + tackles: int = 0 + tackles_won: int = 0 + interceptions: int = 0 + clearances: int = 0 + blocks: int = 0 + aerial_duels_won: int = 0 + aerial_lost: int = 0 + ball_recoveries: int = 0 + dribbled_past: int = 0 + errors_lead_to_goal: int = 0 + errors_lead_to_shot: int = 0 + # Rates recomputed from summed counts, never averaged across competitions + tackles_won_pct: float = 0.0 + aerial_duels_won_pct: float = 0.0 @dataclass diff --git a/backend/app/domain/player_assembler.py b/backend/app/domain/player_assembler.py index e293d2b..747246c 100644 --- a/backend/app/domain/player_assembler.py +++ b/backend/app/domain/player_assembler.py @@ -1,6 +1,7 @@ from datetime import UTC, datetime from app.domain.competitions import canonical_competition +from app.domain.defensive_stats import recompute_rates from app.domain.models import AggregatedScores, CompetitionEntry, PlayerDTO, Stats from app.domain.scoring_engine import ScoringEngine from app.domain.sleeper_detector import SleeperDetector @@ -48,7 +49,19 @@ def aggregate_stats(entries: list[CompetitionEntry]) -> Stats: total.headed_goals += s.headed_goals total.left_foot_goals += s.left_foot_goals total.right_foot_goals += s.right_foot_goals + total.tackles += s.tackles + total.tackles_won += s.tackles_won + total.interceptions += s.interceptions + total.clearances += s.clearances + total.blocks += s.blocks + total.aerial_duels_won += s.aerial_duels_won + total.aerial_lost += s.aerial_lost + total.ball_recoveries += s.ball_recoveries + total.dribbled_past += s.dribbled_past + total.errors_lead_to_goal += s.errors_lead_to_goal + total.errors_lead_to_shot += s.errors_lead_to_shot total.scoring_frequency = (total.minutes / total.goals) if total.goals > 0 else 0.0 + recompute_rates(total) return total diff --git a/backend/app/infrastructure/mongo_repository.py b/backend/app/infrastructure/mongo_repository.py index cee3bc8..3ce17d0 100644 --- a/backend/app/infrastructure/mongo_repository.py +++ b/backend/app/infrastructure/mongo_repository.py @@ -1,3 +1,4 @@ +from dataclasses import asdict, fields from datetime import UTC, datetime from pymongo import ASCENDING, DESCENDING, MongoClient @@ -21,83 +22,13 @@ def _stats_to_dict(stats: Stats) -> dict: - return { - "goals": stats.goals, - "assists": stats.assists, - "xg": stats.xg, - "xa": stats.xa, - "minutes": stats.minutes, - "clean_sheets": stats.clean_sheets, - "pk_saved": stats.pk_saved, - "pk_won": stats.pk_won, - "pk_scored": stats.pk_scored, - "pk_taken": stats.pk_taken, - "yellow_cards": stats.yellow_cards, - "red_cards": stats.red_cards, - "yellow_red_cards": stats.yellow_red_cards, - "direct_red_cards": stats.direct_red_cards, - "fouls_committed": stats.fouls_committed, - "rating": stats.rating, - "big_chances_created": stats.big_chances_created, - "key_passes": stats.key_passes, - "appearances": stats.appearances, - "matches_started": stats.matches_started, - "saves": stats.saves, - "saves_outside_box": stats.saves_outside_box, - "goals_conceded": stats.goals_conceded, - "goals_prevented": stats.goals_prevented, - "high_claims": stats.high_claims, - "penalty_conceded": stats.penalty_conceded, - "penalty_faced": stats.penalty_faced, - "total_shots": stats.total_shots, - "shots_on_target": stats.shots_on_target, - "shots_off_target": stats.shots_off_target, - "scoring_frequency": stats.scoring_frequency, - "penalty_miss": stats.penalty_miss, - "headed_goals": stats.headed_goals, - "left_foot_goals": stats.left_foot_goals, - "right_foot_goals": stats.right_foot_goals, - } + return asdict(stats) def _stats_from_dict(d: dict) -> Stats: - return Stats( - goals=d.get("goals", 0), - assists=d.get("assists", 0), - xg=d.get("xg", 0.0), - xa=d.get("xa", 0.0), - minutes=d.get("minutes", 0), - clean_sheets=d.get("clean_sheets", 0), - pk_saved=d.get("pk_saved", 0), - pk_won=d.get("pk_won", 0), - pk_scored=d.get("pk_scored", 0), - pk_taken=d.get("pk_taken", 0), - yellow_cards=d.get("yellow_cards", 0), - red_cards=d.get("red_cards", 0), - yellow_red_cards=d.get("yellow_red_cards", 0), - direct_red_cards=d.get("direct_red_cards", 0), - fouls_committed=d.get("fouls_committed", 0.0), - rating=d.get("rating", 0.0), - big_chances_created=d.get("big_chances_created", 0), - key_passes=d.get("key_passes", 0), - appearances=d.get("appearances", 0), - matches_started=d.get("matches_started", 0), - saves=d.get("saves", 0), - saves_outside_box=d.get("saves_outside_box", 0), - goals_conceded=d.get("goals_conceded", 0), - goals_prevented=d.get("goals_prevented", 0.0), - high_claims=d.get("high_claims", 0), - penalty_conceded=d.get("penalty_conceded", 0), - penalty_faced=d.get("penalty_faced", 0), - total_shots=d.get("total_shots", 0), - shots_on_target=d.get("shots_on_target", 0), - shots_off_target=d.get("shots_off_target", 0), - scoring_frequency=d.get("scoring_frequency", 0.0), - penalty_miss=d.get("penalty_miss", 0), - headed_goals=d.get("headed_goals", 0), - left_foot_goals=d.get("left_foot_goals", 0), - right_foot_goals=d.get("right_foot_goals", 0), - ) + """Build Stats from a stored doc, ignoring keys the model no longer declares.""" + valid = {f.name for f in fields(Stats)} + return Stats(**{k: v for k, v in d.items() if k in valid}) def _bio_doc(player: PlayerDTO) -> dict: diff --git a/backend/app/modes/fetch_runner.py b/backend/app/modes/fetch_runner.py index ecd7923..6ade321 100644 --- a/backend/app/modes/fetch_runner.py +++ b/backend/app/modes/fetch_runner.py @@ -7,6 +7,7 @@ from app.config import settings from app.domain.competitions import canonical_competition, classify_competition +from app.domain.defensive_stats import apply_defensive_raw from app.domain.models import CompetitionEntry, Stats from app.domain.player_assembler import build_player, merge from app.infrastructure.sofascore_client import SofascoreClient @@ -170,11 +171,13 @@ def _fetch_ss(task_idx: int, comp: str, positions: list[str]) -> None: left_foot_goals=int(row.get("left_foot_goals", 0)), right_foot_goals=int(row.get("right_foot_goals", 0)), ) - position = str(row.get("_position_group") or row.get("position", "MF")) - score = _scoring.calculate(stats, position) raw_stats = row.get("_raw_stats") or {} if not isinstance(raw_stats, dict): raw_stats = {} + apply_defensive_raw(stats, raw_stats) + + position = str(row.get("_position_group") or row.get("position", "MF")) + score = _scoring.calculate(stats, position) canon = canonical_competition(comp) entry = CompetitionEntry( competition=canon, diff --git a/backend/scripts/DB/backfill_defensive_stats.py b/backend/scripts/DB/backfill_defensive_stats.py new file mode 100644 index 0000000..2d903bb --- /dev/null +++ b/backend/scripts/DB/backfill_defensive_stats.py @@ -0,0 +1,98 @@ +"""Backfill the typed defensive Stats fields from each entry's stored raw_stats. + +No re-fetch: the Sofascore defensive columns were always scraped into +competitions[].raw_stats, they were just never promoted into the typed model. Run from +backend/ with Mongo reachable: + + python scripts/DB/backfill_defensive_stats.py # apply + python scripts/DB/backfill_defensive_stats.py --dry-run # report only + +Scores are left untouched: ScoringEngine does not read these fields. +""" + +import os +import sys +from dataclasses import asdict + +from pymongo import MongoClient + +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", ".."))) + +from app.domain.defensive_stats import DEFENSIVE_RAW_MAP, apply_defensive_raw # noqa: E402 +from app.domain.models import Stats # noqa: E402 + +DB_NAME = "football_analytics" + + +def _mongo_uri() -> str: + return os.environ.get("MONGO_URI", "mongodb://localhost:27017/football_analytics") + + +def _stats_from_dict(d: dict) -> Stats: + valid = set(Stats().__dict__) + return Stats(**{k: v for k, v in (d or {}).items() if k in valid}) + + +def _aggregate(entries: list[dict]) -> Stats: + from app.domain.defensive_stats import recompute_rates + + total = Stats() + for entry in entries: + s = _stats_from_dict(entry.get("stats")) + for field in DEFENSIVE_RAW_MAP.values(): + setattr(total, field, getattr(total, field) + getattr(s, field)) + recompute_rates(total) + return total + + +def main() -> None: + dry_run = "--dry-run" in sys.argv + db = MongoClient(_mongo_uri())[DB_NAME] + + scanned = changed = entries_touched = no_raw = 0 + + for doc in db.player_stats.find({}): + scanned += 1 + entries = doc.get("competitions") or [] + updated_entries = [] + doc_changed = False + + for entry in entries: + stats = _stats_from_dict(entry.get("stats")) + raw = entry.get("raw_stats") or {} + if not raw: + no_raw += 1 + before = asdict(stats) + apply_defensive_raw(stats, raw) + if asdict(stats) != before: + doc_changed = True + entries_touched += 1 + entry["stats"] = asdict(stats) + updated_entries.append(entry) + + if not doc_changed: + continue + + agg = _stats_from_dict(doc.get("aggregated_stats")) + totals = _aggregate(updated_entries) + for field in list(DEFENSIVE_RAW_MAP.values()) + [ + "tackles_won_pct", + "aerial_duels_won_pct", + ]: + setattr(agg, field, getattr(totals, field)) + + changed += 1 + if not dry_run: + db.player_stats.update_one( + {"_id": doc["_id"]}, + {"$set": {"competitions": updated_entries, "aggregated_stats": asdict(agg)}}, + ) + + verb = "would update" if dry_run else "updated" + print(f"scanned {scanned} player_stats docs") + print(f"{verb} {changed} docs across {entries_touched} competition entries") + print(f"entries with no raw_stats: {no_raw}") + + +if __name__ == "__main__": + main() diff --git a/backend/tests/domain/test_defensive_stats.py b/backend/tests/domain/test_defensive_stats.py new file mode 100644 index 0000000..686ebb3 --- /dev/null +++ b/backend/tests/domain/test_defensive_stats.py @@ -0,0 +1,164 @@ +import pytest + +from app.domain.defensive_stats import DEFENSIVE_RAW_MAP, apply_defensive_raw, recompute_rates +from app.domain.metric_fields import METRIC_FIELDS +from app.domain.models import CompetitionEntry, Score, Stats +from app.domain.player_assembler import aggregate_stats + +DEFENSIVE_COUNTS = [ + "tackles", + "tackles_won", + "interceptions", + "clearances", + "blocks", + "aerial_duels_won", + "aerial_lost", + "ball_recoveries", + "dribbled_past", + "errors_lead_to_goal", + "errors_lead_to_shot", +] +DEFENSIVE_RATES = ["tackles_won_pct", "aerial_duels_won_pct"] + + +def _entry(competition: str, **stats) -> CompetitionEntry: + return CompetitionEntry( + competition=competition, + stats=Stats(**stats), + scores=Score(offensive=0.0, defensive=0.0, tactical=0.0, s_final=0.0), + raw_stats={}, + total_matches=38, + competition_type="club", + ) + + +def test_stats_declares_every_defensive_field_defaulting_to_zero(): + s = Stats() + for field in DEFENSIVE_COUNTS + DEFENSIVE_RATES: + assert getattr(s, field) == 0, f"{field} should default to 0" + + +def test_defensive_fields_are_queryable_metrics(): + for field in DEFENSIVE_COUNTS + DEFENSIVE_RATES: + assert field in METRIC_FIELDS, f"{field} must be sortable/filterable" + + +def test_apply_defensive_raw_maps_sofascore_columns(): + stats = Stats() + apply_defensive_raw( + stats, + { + "tackles": 46.0, + "tacklesWon": 30.0, + "interceptions": 50.0, + "clearances": 164.0, + "outfielderBlocks": 12.0, + "aerialDuelsWon": 60.0, + "aerialLost": 40.0, + "ballRecovery": 88.0, + "dribbledPast": 21.0, + "errorLeadToGoal": 1.0, + "errorLeadToShot": 3.0, + }, + ) + assert stats.tackles == 46 + assert stats.tackles_won == 30 + assert stats.interceptions == 50 + assert stats.clearances == 164 + assert stats.blocks == 12 + assert stats.aerial_duels_won == 60 + assert stats.aerial_lost == 40 + assert stats.ball_recoveries == 88 + assert stats.dribbled_past == 21 + assert stats.errors_lead_to_goal == 1 + assert stats.errors_lead_to_shot == 3 + + +def test_apply_defensive_raw_tolerates_missing_and_null_columns(): + # GK rows and older documents do not carry every column. + stats = Stats() + apply_defensive_raw(stats, {"tackles": None, "interceptions": 7.0}) + assert stats.tackles == 0 + assert stats.interceptions == 7 + assert stats.clearances == 0 + + +def test_recompute_rates_derives_percentages_from_counts(): + stats = Stats(tackles=100, tackles_won=54, aerial_duels_won=60, aerial_lost=40) + recompute_rates(stats) + assert stats.tackles_won_pct == pytest.approx(54.0) + assert stats.aerial_duels_won_pct == pytest.approx(60.0) + + +def test_recompute_rates_is_zero_when_the_denominator_is_zero(): + stats = Stats(tackles=0, tackles_won=0, aerial_duels_won=0, aerial_lost=0) + recompute_rates(stats) + assert stats.tackles_won_pct == 0.0 + assert stats.aerial_duels_won_pct == 0.0 + + +def test_aggregate_sums_defensive_counts_across_competitions(): + agg = aggregate_stats( + [ + _entry("England Premier League", minutes=900, tackles=46, interceptions=50, blocks=4), + _entry("UEFA Champions League", minutes=450, tackles=14, interceptions=10, blocks=2), + ] + ) + assert agg.tackles == 60 + assert agg.interceptions == 60 + assert agg.blocks == 6 + + +def test_aggregate_recomputes_tackle_success_from_summed_counts_not_averaged(): + # 90% over 10 tackles and 50% over 90 tackles average to 70%, but the true + # combined rate is 54/100 = 54%. Averaging percentages is the bug this pins. + agg = aggregate_stats( + [ + _entry("England Premier League", minutes=900, tackles=10, tackles_won=9), + _entry("UEFA Champions League", minutes=900, tackles=90, tackles_won=45), + ] + ) + assert agg.tackles == 100 + assert agg.tackles_won == 54 + assert agg.tackles_won_pct == pytest.approx(54.0) + + +def test_aggregate_recomputes_aerial_success_from_summed_counts(): + agg = aggregate_stats( + [ + _entry("England Premier League", minutes=900, aerial_duels_won=9, aerial_lost=1), + _entry("UEFA Champions League", minutes=900, aerial_duels_won=45, aerial_lost=45), + ] + ) + assert agg.aerial_duels_won == 54 + assert agg.aerial_lost == 46 + assert agg.aerial_duels_won_pct == pytest.approx(54.0) + + +def test_defensive_raw_map_targets_only_real_stats_fields(): + valid = {f for f in Stats().__dict__} + for raw_key, field in DEFENSIVE_RAW_MAP.items(): + assert field in valid, f"{raw_key} maps to unknown Stats field {field}" + + +def test_defensive_fields_survive_the_mongo_round_trip(): + from app.infrastructure.mongo_repository import _stats_from_dict, _stats_to_dict + + original = Stats(minutes=900, tackles=46, tackles_won=30, interceptions=50, clearances=164) + restored = _stats_from_dict(_stats_to_dict(original)) + assert restored == original + + +def test_stats_from_dict_ignores_unknown_keys_from_older_documents(): + from app.infrastructure.mongo_repository import _stats_from_dict + + restored = _stats_from_dict({"goals": 3, "a_field_we_removed": 99}) + assert restored.goals == 3 + assert restored.tackles == 0 + + +def test_stats_out_exposes_every_stats_field(): + from app.api.modals.player_modals import StatsOut + + missing = set(Stats().__dict__) - set(StatsOut.model_fields) + assert not missing, f"StatsOut silently drops: {sorted(missing)}" diff --git a/docs/superpowers/specs/2026-07-11-defender-representation-design.md b/docs/superpowers/specs/2026-07-11-defender-representation-design.md new file mode 100644 index 0000000..ad86e98 --- /dev/null +++ b/docs/superpowers/specs/2026-07-11-defender-representation-design.md @@ -0,0 +1,154 @@ +# Defender Representation — Design Spec + +**Date:** 2026-07-11 +**Status:** Layer 1 implemented 2026-09-06 on `dev/defender-metrics`. Layer 2 is obsolete. +Layer 3 is still an open proposal. +**Branch:** `dev/defender-metrics` + +## Status update — 2026-09-06 + +**Layer 1 is done.** Thirteen defensive fields are typed on `Stats`, populated from +`raw_stats` on both the live-fetch path and a one-off backfill, aggregated as counts with +the two rates recomputed from the summed counts. They are in `METRIC_FIELDS`, so they are +sortable and filterable, exposed by `StatsOut`, and mirrored in the frontend `Stats` and +`METRIC_OPTIONS`. The backfill updated 1235 of 1256 `player_stats` docs across 1394 +competition entries; no re-fetch was needed, exactly as this spec predicted. + +**Field set — question 1 answered by trimming.** Shipped: `tackles`, `tackles_won`, +`interceptions`, `clearances`, `blocks`, `aerial_duels_won`, `aerial_lost`, +`ball_recoveries`, `dribbled_past`, `errors_lead_to_goal`, `errors_lead_to_shot`, plus +the rates `tackles_won_pct` and `aerial_duels_won_pct`. Dropped for now: `duels_won_pct` +(overlaps the aerial rate) and `pass_accuracy_pct` (a ball-playing metric, not a +defending one). Both remain available in `raw_stats` if wanted later. + +**Layer 2 is obsolete.** It targeted the RAG document builder +(`backend/app/domain/player_document.py`) for embedding quality. RAG was cancelled and +that module no longer exists. The consumer of Layer 1 is now the Rankings filters and the +chatbot agent's `defending` tool, which read `METRIC_FIELDS` directly — no prose needed. + +**Layer 3 is unchanged and still needs sign-off.** Scoring was deliberately left out: +`ScoringEngine` does not read these fields, so `s_final` is untouched and no +`Mathematical_Specification.md` update is required by this change. Questions 3 and 4 +below are still open. + +## Problem + +Defenders are currently evaluated and described through an **attacking lens**. Both +the RAG document builder (`playground/document_builder.py` → `backend/app/domain/player_document.py`) +and the `s_final` scoring see only the typed `Stats` model, which is an attacking + +goalkeeping schema. Its only defensive signal is `clean_sheets` (a team-level proxy) +and `fouls_committed`. + +Concretely, `_outfield_headline()` picks a defender's adjective from *finishing* and +*creativity* traits only; when neither fires it emits the static string +`"defensively solid"` with **no data behind it**. Result: semantic retrieval for +queries like *"solid defender"* ranks players on attacking prose, and star defenders +(whose docs emphasise their goals/assists) are mis-ranked. + +## Data availability — the metrics already exist + +Every defensive stat we want is **already scraped and stored** in each competition +entry's untyped `raw_stats` blob; we simply never promote it into the typed model. +Verified present on real player docs: + +| Concept | `raw_stats` key | +|---|---| +| Tackles / won / win % | `tackles`, `tacklesWon`, `tacklesWonPercentage` | +| Interceptions | `interceptions` | +| Clearances | `clearances` | +| Blocks | `outfielderBlocks`, `blockedShots` | +| Aerial duels won / win % | `aerialDuelsWon`, `aerialDuelsWonPercentage`, `aerialLost` | +| Ground/total duels win % | `groundDuelsWonPercentage`, `totalDuelsWonPercentage` | +| Ball recoveries | `ballRecovery` | +| Dribbled past (negative) | `dribbledPast` | +| Errors (negative) | `errorLeadToGoal`, `errorLeadToShot` | +| Passing quality (ball-players) | `accuratePassesPercentage`, `accurateLongBallsPercentage` | + +No new scraping is required. + +## Layer 1 — Promote defensive fields into `Stats` + +Add these fields to `Stats` in `backend/app/domain/models.py` and populate them in +`player_assembler` from `raw_stats` (defaulting to 0 when absent — GK/older rows): + +- Counts: `tackles`, `tackles_won`, `interceptions`, `clearances`, `blocks` + (`outfielderBlocks`), `aerial_duels_won`, `aerial_lost`, `ball_recoveries`, + `dribbled_past`, `errors_lead_to_goal`, `errors_lead_to_shot` +- Percentages: `tackles_won_pct`, `aerial_duels_won_pct`, `duels_won_pct` + (`totalDuelsWonPercentage`), `pass_accuracy_pct` + +**Aggregation rule (important):** percentages must be **recomputed from summed +counts** across competitions (e.g. `tackles_won_pct = Σtackles_won / Σtackles`), NOT +averaged — averaging percentages across competitions with different volumes is wrong. +Where a raw count for the denominator is unavailable, keep the season percentage as-is +and document the limitation. + +Mirror the new fields into `StatsOut` (API) and `Stats` (frontend `players.ts`) and add +the useful ones to `METRIC_OPTIONS` / `METRIC_FIELDS` so they become sortable/filterable +(this also makes them answerable on the structured RAG path — e.g. "best tacklers"). + +## Layer 2 — Defender-aware document (OBSOLETE — RAG cancelled) + +Add `_defender_headline(player)` and a defender branch in the summary. Priority order +matches the analyst's mental model: **clean sheets → tackle success → aerial dominance +→ interceptions/clearances → then goals/assists.** + +**Clean sheets — a representation gap, not a data gap.** `clean_sheets` is already +captured (Sofascore `cleanSheet` → typed `Stats`, aggregated, and scored `×4` for DF), +but the document builder currently mentions it **only in the per-competition stat line** — +the outfield *summary prose* and the *aggregated per-90 block* omit it entirely. So the +embedding barely sees it. Surfacing it is part of this layer: clean sheets must appear in +the defender **headline** and in the aggregated defensive line, not just per-competition. + +Proposed headline adjective (first match wins): +- `"commanding"` — high `aerial_duels_won_pct` (≥ 60%) with meaningful clean sheets +- `"combative ball-winner"` — high `tackles_won_pct` (≥ 65%) and interceptions per 90 +- `"ball-playing"` — high `pass_accuracy_pct` (≥ 88%) / accurate long balls +- `"dependable"` — otherwise (replaces the empty `"defensively solid"`) + +Add a **defensive stat line** to the outfield document (populated for all outfielders, +but it is the *lead* for DF): +``` +Defensive: 2.3 tackles/90 (71% won), 1.8 interceptions/90, 3.1 clearances/90, + 68% aerial duels won, 0.4 blocks/90, 0 errors leading to a goal. +``` +Attacking output stays, but for DF it moves **after** the defensive line. + +## Layer 3 — Scoring (PROPOSAL — still needs sign-off, not implemented) + +Optional; only if we choose "docs + scoring". Fold defensive output into `s_final` for +`position == "DF"` (and partially MF). This changes app-wide ranking and requires a +`Mathematical_Specification.md` update. **All weights below are placeholders for review, +not final:** + +| Metric (per 90) | Proposed weight | +|---|---| +| Tackle won | +0.6 | +| Interception | +0.6 | +| Clearance | +0.2 | +| Block | +0.5 | +| Aerial duel won | +0.4 | +| Ball recovery | +0.1 | +| `errorLeadToGoal` | −3.0 | +| `errorLeadToShot` | −1.0 | +| Dribbled past | −0.3 | + +Quality percentages (`tackles_won_pct`, `aerial_duels_won_pct`) act as **multipliers** +on the corresponding volume term rather than standalone additive points, so high-volume +low-quality defenders aren't overrated. Normalisation and `playing_time_factor` / +`starter_bonus` stay as in the current `s_final`. + +## Out of scope + +- No new scraping or data sources. +- No change to GK scoring/representation. +- No re-fetch required — a one-off backfill re-reads existing `raw_stats` into the new + typed fields (a small migration script over `player_stats`). + +## Open questions for review + +1. Field set — is the Layer-1 list right, or trim/extend (e.g. drop `duels_won_pct`)? +2. Headline thresholds (60% aerial, 65% tackle, 88% pass) — reasonable, or tune? +3. Scoring: in scope now, or docs+retrieval only and defer scoring to a follow-up? +4. If scoring is in scope, do the proposed weights need a data-distribution check first + (data-analyst) before we commit numbers to the math spec? diff --git a/frontend/src/api/players.ts b/frontend/src/api/players.ts index a0a6b53..28f4beb 100644 --- a/frontend/src/api/players.ts +++ b/frontend/src/api/players.ts @@ -13,6 +13,11 @@ export interface Stats { total_shots: number; shots_on_target: number; shots_off_target: number scoring_frequency: number; penalty_miss: number headed_goals: number; left_foot_goals: number; right_foot_goals: number + tackles: number; tackles_won: number; tackles_won_pct: number + interceptions: number; clearances: number; blocks: number + aerial_duels_won: number; aerial_lost: number; aerial_duels_won_pct: number + ball_recoveries: number; dribbled_past: number + errors_lead_to_goal: number; errors_lead_to_shot: number } export interface Score { @@ -78,6 +83,18 @@ export const METRIC_OPTIONS: { value: string; label: string }[] = [ { value: 'pk_saved', label: 'Penalties saved' }, { value: 'penalty_miss', label: 'Penalties missed' }, { value: 'penalty_conceded', label: 'Penalties conceded' }, + { value: 'tackles', label: 'Tackles' }, + { value: 'tackles_won', label: 'Tackles won' }, + { value: 'tackles_won_pct', label: 'Tackle success %' }, + { value: 'interceptions', label: 'Interceptions' }, + { value: 'clearances', label: 'Clearances' }, + { value: 'blocks', label: 'Blocks' }, + { value: 'aerial_duels_won', label: 'Aerial duels won' }, + { value: 'aerial_duels_won_pct', label: 'Aerial duel success %' }, + { value: 'ball_recoveries', label: 'Ball recoveries' }, + { value: 'dribbled_past', label: 'Dribbled past' }, + { value: 'errors_lead_to_goal', label: 'Errors leading to a goal' }, + { value: 'errors_lead_to_shot', label: 'Errors leading to a shot' }, { value: 'penalty_faced', label: 'Penalties faced' }, { value: 'yellow_red_cards', label: 'Second-yellow reds' }, { value: 'direct_red_cards', label: 'Direct reds' }, From 342e70aed52dcaaeb0199fbec6aa02dc21663c67 Mon Sep 17 00:00:00 2001 From: Yonatan Hen Date: Sun, 6 Sep 2026 12:19:28 +0300 Subject: [PATCH 3/5] docs: document the defensive metrics and correct the stale S_final formula The Defensive Metrics bullet is the user-visible half of the previous commit: 13 metrics reachable through METRIC_FIELDS and selectable in the Rankings metric picker. It says plainly that scoring is untouched, so it cannot be misread as a scoring change. The S_final correction is unrelated to this feature and pre-existing: the README still described the pre-overhaul formula, in both the Core Features bullet and the architecture diagram, after PR #28 replaced it. Fixed here because the new bullet refers to S_final and would otherwise sit under a wrong definition of it. Mathematical_Specification.md deliberately unchanged: ScoringEngine reads none of the new fields, Defensive_DF is still CS x 4, and nothing in that document became inaccurate. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SKAXmoCoJ3gAu4yyD3zkCL --- README.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 2c4fb03..420c7c8 100644 --- a/README.md +++ b/README.md @@ -37,7 +37,7 @@ flowchart LR FE["React SPA\nVite :5173"] subgraph Backend["FastAPI :8000"] API["API Routers"] - SE["Scoring Engine\nS_final = (Off + Def + Tac) / (min/90)"] + SE["Scoring Engine\nS_final = raw/90 x starter x confidence + bonus"] PA["Player Assembler\nbuild · merge · aggregate"] SC["Stats Client\nScraperFC + Chrome"] MR["Mongo Repository"] @@ -65,8 +65,9 @@ flowchart LR ## Core Features -- **Fantasy Scoring** — composite score `S_final = (Offensive + Defensive + Tactical) / (minutes / 90)` with position-specific goal/assist weights; GK goals worth 10 pts, FW goals worth 4 pts +- **Fantasy Scoring** — composite score `S_final = raw_per90 x starter_bonus x confidence + playing_time_bonus`, where `raw_per90` is `(Offensive + Defensive + Tactical) / (minutes / 90)` with position-specific goal/assist weights (GK goals worth 10 pts, FW goals worth 4 pts). `starter_bonus` rewards regular starters and `confidence` discounts small appearance counts — see `Mathematical_Specification.md` - **Rankings** — paginated player table sorted by `S_final`; filterable by position, team, nationality, and sleeper flag +- **Defensive Metrics** — tackles, interceptions, clearances, blocks, aerial duels, ball recoveries, and errors leading to a shot/goal are tracked per player and sortable/filterable in Rankings; not yet part of `S_final` scoring - **Sleeper Detection** — `HIGH_VALUE` flags players where xG+xA significantly exceeds G+A; `OVERPERFORMING` flags the inverse; gated on `minutes > 450` - **Player Detail** — per-competition stat breakdown and aggregated scores for any player, including those without a linked external ID - **Head-to-Head Compare** — side-by-side comparison of exactly two players across all stat dimensions From 9ed4b82a138cd6d73b3dd8821d1b4a7979c56a03 Mon Sep 17 00:00:00 2001 From: Yonatan Hen Date: Sun, 6 Sep 2026 12:27:14 +0300 Subject: [PATCH 4/5] =?UTF-8?q?fix:=20address=20code=20review=20=E2=80=94?= =?UTF-8?q?=20backfill=20coverage,=20restored=20PR=20gate,=20frontend=20mi?= =?UTF-8?q?rror?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Restores the PreToolUse /code-review gate. Commit 43f4f90 carried the settings file over from dev/chatbot-agent and deleted that whole block while its message claimed only to rescope the doc hook. Merging it would have removed the code-review gate from master as a side effect of a metrics PR. Both hooks now use the Bash(gh pr create:*) permission-prefix form, matching the allow rules in the same file; Bash(gh pr create *) would not have matched a bare `gh pr create` with no flags. Backfill now writes every scanned doc, not only changed ones. 21 docs had all-zero defensive columns, so they were skipped and kept no defensive keys at all. Mongo's $gte does not match a missing field, so those players disappeared from any defensive filter through get_players, while the stats_view path returned them because _stats_from_dict defaults them to 0 in Python. One query, two answers. Re-run: all 1256 docs now match a tackles >= 0 filter, previously 1235. aerial_lost was typed and allowlisted but had no METRIC_OPTIONS entry, so it could not be filtered from the UI, and the spec wrongly claimed the fields were mirrored. Added, and the defensive block moved out of the middle of the goalkeeping group. A new test compares METRIC_OPTIONS against METRIC_FIELDS so this drift fails the suite next time. Also documents that fetches must be stopped before running the backfill, which rewrites each doc's whole competitions array. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SKAXmoCoJ3gAu4yyD3zkCL --- .claude/settings.json | 16 ++++++++++++++-- backend/scripts/DB/backfill_defensive_stats.py | 15 ++++++++++----- backend/tests/domain/test_defensive_stats.py | 16 ++++++++++++++++ frontend/src/api/players.ts | 15 ++++++++------- 4 files changed, 48 insertions(+), 14 deletions(-) diff --git a/.claude/settings.json b/.claude/settings.json index 645742f..1eb0f68 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -20,17 +20,29 @@ ] }, "hooks": { + "PreToolUse": [ + { + "matcher": "Bash", + "hooks": [ + { + "type": "prompt", + "if": "Bash(gh pr create:*)", + "prompt": "Hook input: $ARGUMENTS\n\nFirst, read tool_input.command. If it does not invoke `gh pr create`, respond {\"ok\": true} immediately and do nothing else.\n\nOtherwise a PR to master is about to be opened. If the code-review skill (/code-review) has NOT already been run against the latest changes in this conversation, respond {\"ok\": false, \"reason\": \"Run /code-review, address its findings, then retry opening the PR.\"}. Otherwise respond {\"ok\": true}." + } + ] + } + ], "PostToolUse": [ { "matcher": "Bash", "hooks": [ { "type": "prompt", - "if": "Bash(gh pr create *)", + "if": "Bash(gh pr create:*)", "prompt": "Hook input: $ARGUMENTS\n\nFirst, read tool_input.command. If it does not invoke `gh pr create`, respond {\"ok\": true} immediately and do nothing else.\n\nOtherwise a PR to master was just opened. Unless the technical-writer subagent has already checked docs against this diff earlier in this conversation, respond {\"ok\": false, \"reason\": \"Invoke the technical-writer subagent (.claude/agents/technical-writer.md), make sure you document all changes in the PR summary correctly, and the README.md + Mathematical_Specification.md files only if necessary.\"}. Otherwise respond {\"ok\": true}." } ] } ] } -} \ No newline at end of file +} diff --git a/backend/scripts/DB/backfill_defensive_stats.py b/backend/scripts/DB/backfill_defensive_stats.py index 2d903bb..7fec061 100644 --- a/backend/scripts/DB/backfill_defensive_stats.py +++ b/backend/scripts/DB/backfill_defensive_stats.py @@ -8,6 +8,9 @@ python scripts/DB/backfill_defensive_stats.py --dry-run # report only Scores are left untouched: ScoringEngine does not read these fields. + +Stop any running fetch job first: this rewrites each doc's whole competitions array, so a +concurrent fetch writing between the read and the write would be overwritten. """ import os @@ -70,9 +73,10 @@ def main() -> None: entry["stats"] = asdict(stats) updated_entries.append(entry) - if not doc_changed: - continue - + # Write every doc, not only changed ones. A doc left without the new keys is not + # merely stale: Mongo's $gte does not match a missing field, so those players + # disappear from defensive filters, while the stats_view path (which defaults them + # to 0 in Python) still returns them. Same query, two answers. agg = _stats_from_dict(doc.get("aggregated_stats")) totals = _aggregate(updated_entries) for field in list(DEFENSIVE_RAW_MAP.values()) + [ @@ -81,16 +85,17 @@ def main() -> None: ]: setattr(agg, field, getattr(totals, field)) - changed += 1 + changed += 1 if doc_changed else 0 if not dry_run: db.player_stats.update_one( {"_id": doc["_id"]}, {"$set": {"competitions": updated_entries, "aggregated_stats": asdict(agg)}}, ) - verb = "would update" if dry_run else "updated" + verb = "would change" if dry_run else "changed" print(f"scanned {scanned} player_stats docs") print(f"{verb} {changed} docs across {entries_touched} competition entries") + print("every scanned doc was written, so no doc is left without the new keys") print(f"entries with no raw_stats: {no_raw}") diff --git a/backend/tests/domain/test_defensive_stats.py b/backend/tests/domain/test_defensive_stats.py index 686ebb3..9f4e40b 100644 --- a/backend/tests/domain/test_defensive_stats.py +++ b/backend/tests/domain/test_defensive_stats.py @@ -162,3 +162,19 @@ def test_stats_out_exposes_every_stats_field(): missing = set(Stats().__dict__) - set(StatsOut.model_fields) assert not missing, f"StatsOut silently drops: {sorted(missing)}" + + +def test_frontend_metric_options_mirror_the_backend_allowlist(): + # players.ts claims to mirror METRIC_FIELDS. aerial_lost was typed and allowlisted + # but had no dropdown entry, so it could not be filtered from the UI. + import re + from pathlib import Path + + players_ts = Path(__file__).resolve().parents[2].parent / "frontend/src/api/players.ts" + after = players_ts.read_text(encoding="utf-8").split("METRIC_OPTIONS")[1] + block = after[: after.index("\n]")] + options = set(re.findall(r"\{ value: '([a-z_]+)'", block)) + assert options == set(METRIC_FIELDS), ( + f"only in backend: {sorted(set(METRIC_FIELDS) - options)}; " + f"only in frontend: {sorted(options - set(METRIC_FIELDS))}" + ) diff --git a/frontend/src/api/players.ts b/frontend/src/api/players.ts index 28f4beb..72febc0 100644 --- a/frontend/src/api/players.ts +++ b/frontend/src/api/players.ts @@ -83,6 +83,13 @@ export const METRIC_OPTIONS: { value: string; label: string }[] = [ { value: 'pk_saved', label: 'Penalties saved' }, { value: 'penalty_miss', label: 'Penalties missed' }, { value: 'penalty_conceded', label: 'Penalties conceded' }, + { value: 'penalty_faced', label: 'Penalties faced' }, + { value: 'yellow_red_cards', label: 'Second-yellow reds' }, + { value: 'direct_red_cards', label: 'Direct reds' }, + { value: 'scoring_frequency', label: 'Scoring frequency' }, + { value: 'headed_goals', label: 'Headed goals' }, + { value: 'left_foot_goals', label: 'Left-foot goals' }, + { value: 'right_foot_goals', label: 'Right-foot goals' }, { value: 'tackles', label: 'Tackles' }, { value: 'tackles_won', label: 'Tackles won' }, { value: 'tackles_won_pct', label: 'Tackle success %' }, @@ -90,18 +97,12 @@ export const METRIC_OPTIONS: { value: string; label: string }[] = [ { value: 'clearances', label: 'Clearances' }, { value: 'blocks', label: 'Blocks' }, { value: 'aerial_duels_won', label: 'Aerial duels won' }, + { value: 'aerial_lost', label: 'Aerial duels lost' }, { value: 'aerial_duels_won_pct', label: 'Aerial duel success %' }, { value: 'ball_recoveries', label: 'Ball recoveries' }, { value: 'dribbled_past', label: 'Dribbled past' }, { value: 'errors_lead_to_goal', label: 'Errors leading to a goal' }, { value: 'errors_lead_to_shot', label: 'Errors leading to a shot' }, - { value: 'penalty_faced', label: 'Penalties faced' }, - { value: 'yellow_red_cards', label: 'Second-yellow reds' }, - { value: 'direct_red_cards', label: 'Direct reds' }, - { value: 'scoring_frequency', label: 'Scoring frequency' }, - { value: 'headed_goals', label: 'Headed goals' }, - { value: 'left_foot_goals', label: 'Left-foot goals' }, - { value: 'right_foot_goals', label: 'Right-foot goals' }, { value: 'offensive', label: 'Offensive score' }, { value: 'defensive', label: 'Defensive score' }, { value: 'tactical', label: 'Tactical score' }, From 1df146450cb92ac3fc0b5f13a3aba15e95e0018e Mon Sep 17 00:00:00 2001 From: Yonatan Hen Date: Sun, 6 Sep 2026 13:12:29 +0300 Subject: [PATCH 5/5] =?UTF-8?q?fix:=20second=20review=20pass=20=E2=80=94?= =?UTF-8?q?=20container-safe=20test,=20CI=20filter,=20honest=20dry-run?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - The frontend mirror test read frontend/src/api/players.ts by relative path. docker-compose mounts only ./backend at /app, so in the container that path resolved outside the mount and the test raised FileNotFoundError. It now skips when the file is absent, so a backend-only checkout or container run is unaffected. - The backend CI job is gated on a backend/** paths filter, so a PR touching only players.ts skipped it and the mirror test never ran — exactly the drift it exists to catch. players.ts is now in that filter. - The backfill printed "every scanned doc was written" unconditionally, including under --dry-run when nothing was written. That is the same kind of false reassurance the change was meant to remove. Dry-run and real runs now report separately, with their own written counter. - The METRIC_OPTIONS extractor now asserts it parsed something, so a formatting change fails as "could not parse" instead of a misleading field mismatch, and it accepts digits in metric names. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01SKAXmoCoJ3gAu4yyD3zkCL --- .github/workflows/ci.yml | 2 ++ backend/scripts/DB/backfill_defensive_stats.py | 8 ++++++-- backend/tests/domain/test_defensive_stats.py | 5 ++++- 3 files changed, 12 insertions(+), 3 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3566031..51a4dbd 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -26,6 +26,8 @@ jobs: backend: - 'backend/**' - '.github/workflows/ci.yml' + # mirrored against METRIC_FIELDS by tests/domain/test_defensive_stats.py + - 'frontend/src/api/players.ts' frontend: - 'frontend/**' - '.github/workflows/ci.yml' diff --git a/backend/scripts/DB/backfill_defensive_stats.py b/backend/scripts/DB/backfill_defensive_stats.py index 7fec061..3b79897 100644 --- a/backend/scripts/DB/backfill_defensive_stats.py +++ b/backend/scripts/DB/backfill_defensive_stats.py @@ -52,7 +52,7 @@ def main() -> None: dry_run = "--dry-run" in sys.argv db = MongoClient(_mongo_uri())[DB_NAME] - scanned = changed = entries_touched = no_raw = 0 + scanned = changed = written = entries_touched = no_raw = 0 for doc in db.player_stats.find({}): scanned += 1 @@ -87,6 +87,7 @@ def main() -> None: changed += 1 if doc_changed else 0 if not dry_run: + written += 1 db.player_stats.update_one( {"_id": doc["_id"]}, {"$set": {"competitions": updated_entries, "aggregated_stats": asdict(agg)}}, @@ -95,7 +96,10 @@ def main() -> None: verb = "would change" if dry_run else "changed" print(f"scanned {scanned} player_stats docs") print(f"{verb} {changed} docs across {entries_touched} competition entries") - print("every scanned doc was written, so no doc is left without the new keys") + if dry_run: + print(f"would write all {scanned} docs (nothing was written: --dry-run)") + else: + print(f"wrote {written} docs, so none is left without the new keys") print(f"entries with no raw_stats: {no_raw}") diff --git a/backend/tests/domain/test_defensive_stats.py b/backend/tests/domain/test_defensive_stats.py index 9f4e40b..c548a00 100644 --- a/backend/tests/domain/test_defensive_stats.py +++ b/backend/tests/domain/test_defensive_stats.py @@ -171,9 +171,12 @@ def test_frontend_metric_options_mirror_the_backend_allowlist(): from pathlib import Path players_ts = Path(__file__).resolve().parents[2].parent / "frontend/src/api/players.ts" + if not players_ts.is_file(): + pytest.skip("frontend not present (backend-only checkout or container)") after = players_ts.read_text(encoding="utf-8").split("METRIC_OPTIONS")[1] block = after[: after.index("\n]")] - options = set(re.findall(r"\{ value: '([a-z_]+)'", block)) + options = set(re.findall(r"\{ value: '([a-z_0-9]+)'", block)) + assert options, "could not parse METRIC_OPTIONS; the extractor needs updating" assert options == set(METRIC_FIELDS), ( f"only in backend: {sorted(set(METRIC_FIELDS) - options)}; " f"only in frontend: {sorted(options - set(METRIC_FIELDS))}"