diff --git a/.claude/settings.json b/.claude/settings.json index 496c9a7..1eb0f68 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -3,6 +3,22 @@ "CLAUDE_CODE_EXPERIMENTAL_AGENT_TEAMS": "1" }, "teammateMode": "in-process", + "permissions": { + "allow": [ + "Bash(pytest:*)", + "Bash(python -m pytest:*)", + "Bash(.venv/Scripts/python -m pytest:*)", + "Bash(.venv\\Scripts\\python -m pytest:*)", + "Bash(backend/.venv/Scripts/python -m pytest:*)", + "Bash(backend\\.venv\\Scripts\\python -m pytest:*)", + "Bash(ruff:*)", + "Bash(python -m ruff:*)", + "Bash(.venv/Scripts/python -m ruff:*)", + "Bash(.venv\\Scripts\\python -m ruff:*)", + "Bash(backend/.venv/Scripts/python -m ruff:*)", + "Bash(backend\\.venv\\Scripts\\python -m ruff:*)" + ] + }, "hooks": { "PreToolUse": [ { @@ -10,8 +26,8 @@ "hooks": [ { "type": "prompt", - "if": "Bash(gh pr create*)", - "prompt": "A PR to master is about to be opened. If the code-review skill (/code-review) has NOT already been run against the latest changes in this conversation, respond {\"ok\": false, \"reason\": \"Run /code-review, address its findings, then retry opening the PR.\"}. Otherwise respond {\"ok\": true}." + "if": "Bash(gh pr create:*)", + "prompt": "Hook input: $ARGUMENTS\n\nFirst, read tool_input.command. If it does not invoke `gh pr create`, respond {\"ok\": true} immediately and do nothing else.\n\nOtherwise a PR to master is about to be opened. If the code-review skill (/code-review) has NOT already been run against the latest changes in this conversation, respond {\"ok\": false, \"reason\": \"Run /code-review, address its findings, then retry opening the PR.\"}. Otherwise respond {\"ok\": true}." } ] } @@ -22,11 +38,11 @@ "hooks": [ { "type": "prompt", - "if": "Bash(gh pr create*)", - "prompt": "A PR to master was just opened. Unless the technical-writer subagent has already checked docs against this diff earlier in this conversation, respond {\"ok\": false, \"reason\": \"Invoke the technical-writer subagent (.claude/agents/technical-writer.md), make sure you document all changes in the PR summary corrrecly, and the README.md + Mathematical_Specification.md files only if necessary.\"}. Otherwise respond {\"ok\": true}." + "if": "Bash(gh pr create:*)", + "prompt": "Hook input: $ARGUMENTS\n\nFirst, read tool_input.command. If it does not invoke `gh pr create`, respond {\"ok\": true} immediately and do nothing else.\n\nOtherwise a PR to master was just opened. Unless the technical-writer subagent has already checked docs against this diff earlier in this conversation, respond {\"ok\": false, \"reason\": \"Invoke the technical-writer subagent (.claude/agents/technical-writer.md), make sure you document all changes in the PR summary correctly, and the README.md + Mathematical_Specification.md files only if necessary.\"}. Otherwise respond {\"ok\": true}." } ] } ] } -} \ No newline at end of file +} diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3566031..51a4dbd 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -26,6 +26,8 @@ jobs: backend: - 'backend/**' - '.github/workflows/ci.yml' + # mirrored against METRIC_FIELDS by tests/domain/test_defensive_stats.py + - 'frontend/src/api/players.ts' frontend: - 'frontend/**' - '.github/workflows/ci.yml' diff --git a/README.md b/README.md index 2c4fb03..420c7c8 100644 --- a/README.md +++ b/README.md @@ -37,7 +37,7 @@ flowchart LR FE["React SPA\nVite :5173"] subgraph Backend["FastAPI :8000"] API["API Routers"] - SE["Scoring Engine\nS_final = (Off + Def + Tac) / (min/90)"] + SE["Scoring Engine\nS_final = raw/90 x starter x confidence + bonus"] PA["Player Assembler\nbuild · merge · aggregate"] SC["Stats Client\nScraperFC + Chrome"] MR["Mongo Repository"] @@ -65,8 +65,9 @@ flowchart LR ## Core Features -- **Fantasy Scoring** — composite score `S_final = (Offensive + Defensive + Tactical) / (minutes / 90)` with position-specific goal/assist weights; GK goals worth 10 pts, FW goals worth 4 pts +- **Fantasy Scoring** — composite score `S_final = raw_per90 x starter_bonus x confidence + playing_time_bonus`, where `raw_per90` is `(Offensive + Defensive + Tactical) / (minutes / 90)` with position-specific goal/assist weights (GK goals worth 10 pts, FW goals worth 4 pts). `starter_bonus` rewards regular starters and `confidence` discounts small appearance counts — see `Mathematical_Specification.md` - **Rankings** — paginated player table sorted by `S_final`; filterable by position, team, nationality, and sleeper flag +- **Defensive Metrics** — tackles, interceptions, clearances, blocks, aerial duels, ball recoveries, and errors leading to a shot/goal are tracked per player and sortable/filterable in Rankings; not yet part of `S_final` scoring - **Sleeper Detection** — `HIGH_VALUE` flags players where xG+xA significantly exceeds G+A; `OVERPERFORMING` flags the inverse; gated on `minutes > 450` - **Player Detail** — per-competition stat breakdown and aggregated scores for any player, including those without a linked external ID - **Head-to-Head Compare** — side-by-side comparison of exactly two players across all stat dimensions diff --git a/backend/app/api/modals/player_modals.py b/backend/app/api/modals/player_modals.py index 3fc3937..04a6345 100644 --- a/backend/app/api/modals/player_modals.py +++ b/backend/app/api/modals/player_modals.py @@ -39,6 +39,20 @@ class StatsOut(BaseModel): headed_goals: int left_foot_goals: int right_foot_goals: int + # Defending + tackles: int = 0 + tackles_won: int = 0 + interceptions: int = 0 + clearances: int = 0 + blocks: int = 0 + aerial_duels_won: int = 0 + aerial_lost: int = 0 + ball_recoveries: int = 0 + dribbled_past: int = 0 + errors_lead_to_goal: int = 0 + errors_lead_to_shot: int = 0 + tackles_won_pct: float = 0.0 + aerial_duels_won_pct: float = 0.0 class ScoreOut(BaseModel): diff --git a/backend/app/domain/defensive_stats.py b/backend/app/domain/defensive_stats.py new file mode 100644 index 0000000..e22d935 --- /dev/null +++ b/backend/app/domain/defensive_stats.py @@ -0,0 +1,49 @@ +"""Promote Sofascore defensive columns from raw_stats into the typed Stats model. + +raw_stats is the single source for these, so a live fetch and the backfill migration +share one mapping. Rates are derived from counts, never read from Sofascore's own +percentage columns, so they stay correct once counts are summed across competitions. +""" + +from app.domain.models import Stats + +DEFENSIVE_RAW_MAP: dict[str, str] = { + "tackles": "tackles", + "tacklesWon": "tackles_won", + "interceptions": "interceptions", + "clearances": "clearances", + "outfielderBlocks": "blocks", + "aerialDuelsWon": "aerial_duels_won", + "aerialLost": "aerial_lost", + "ballRecovery": "ball_recoveries", + "dribbledPast": "dribbled_past", + "errorLeadToGoal": "errors_lead_to_goal", + "errorLeadToShot": "errors_lead_to_shot", +} + + +def apply_defensive_raw(stats: Stats, raw: dict | None) -> Stats: + """Copy the defensive columns out of one entry's raw_stats onto stats, then derive rates.""" + for raw_key, field in DEFENSIVE_RAW_MAP.items(): + value = (raw or {}).get(raw_key) + if value is None: + continue + try: + setattr(stats, field, int(float(value))) + except (TypeError, ValueError): + continue + recompute_rates(stats) + return stats + + +def recompute_rates(stats: Stats) -> Stats: + """Derive success rates from the current counts. Safe to call after aggregation.""" + stats.tackles_won_pct = _pct(stats.tackles_won, stats.tackles) + stats.aerial_duels_won_pct = _pct( + stats.aerial_duels_won, stats.aerial_duels_won + stats.aerial_lost + ) + return stats + + +def _pct(won: float, attempted: float) -> float: + return round(won / attempted * 100, 1) if attempted > 0 else 0.0 diff --git a/backend/app/domain/models.py b/backend/app/domain/models.py index c8c5acb..99853db 100644 --- a/backend/app/domain/models.py +++ b/backend/app/domain/models.py @@ -42,6 +42,21 @@ class Stats: headed_goals: int = 0 left_foot_goals: int = 0 right_foot_goals: int = 0 + # Defending (promoted from raw_stats; see defensive_stats.py) + tackles: int = 0 + tackles_won: int = 0 + interceptions: int = 0 + clearances: int = 0 + blocks: int = 0 + aerial_duels_won: int = 0 + aerial_lost: int = 0 + ball_recoveries: int = 0 + dribbled_past: int = 0 + errors_lead_to_goal: int = 0 + errors_lead_to_shot: int = 0 + # Rates recomputed from summed counts, never averaged across competitions + tackles_won_pct: float = 0.0 + aerial_duels_won_pct: float = 0.0 @dataclass diff --git a/backend/app/domain/player_assembler.py b/backend/app/domain/player_assembler.py index e293d2b..747246c 100644 --- a/backend/app/domain/player_assembler.py +++ b/backend/app/domain/player_assembler.py @@ -1,6 +1,7 @@ from datetime import UTC, datetime from app.domain.competitions import canonical_competition +from app.domain.defensive_stats import recompute_rates from app.domain.models import AggregatedScores, CompetitionEntry, PlayerDTO, Stats from app.domain.scoring_engine import ScoringEngine from app.domain.sleeper_detector import SleeperDetector @@ -48,7 +49,19 @@ def aggregate_stats(entries: list[CompetitionEntry]) -> Stats: total.headed_goals += s.headed_goals total.left_foot_goals += s.left_foot_goals total.right_foot_goals += s.right_foot_goals + total.tackles += s.tackles + total.tackles_won += s.tackles_won + total.interceptions += s.interceptions + total.clearances += s.clearances + total.blocks += s.blocks + total.aerial_duels_won += s.aerial_duels_won + total.aerial_lost += s.aerial_lost + total.ball_recoveries += s.ball_recoveries + total.dribbled_past += s.dribbled_past + total.errors_lead_to_goal += s.errors_lead_to_goal + total.errors_lead_to_shot += s.errors_lead_to_shot total.scoring_frequency = (total.minutes / total.goals) if total.goals > 0 else 0.0 + recompute_rates(total) return total diff --git a/backend/app/infrastructure/mongo_repository.py b/backend/app/infrastructure/mongo_repository.py index cee3bc8..3ce17d0 100644 --- a/backend/app/infrastructure/mongo_repository.py +++ b/backend/app/infrastructure/mongo_repository.py @@ -1,3 +1,4 @@ +from dataclasses import asdict, fields from datetime import UTC, datetime from pymongo import ASCENDING, DESCENDING, MongoClient @@ -21,83 +22,13 @@ def _stats_to_dict(stats: Stats) -> dict: - return { - "goals": stats.goals, - "assists": stats.assists, - "xg": stats.xg, - "xa": stats.xa, - "minutes": stats.minutes, - "clean_sheets": stats.clean_sheets, - "pk_saved": stats.pk_saved, - "pk_won": stats.pk_won, - "pk_scored": stats.pk_scored, - "pk_taken": stats.pk_taken, - "yellow_cards": stats.yellow_cards, - "red_cards": stats.red_cards, - "yellow_red_cards": stats.yellow_red_cards, - "direct_red_cards": stats.direct_red_cards, - "fouls_committed": stats.fouls_committed, - "rating": stats.rating, - "big_chances_created": stats.big_chances_created, - "key_passes": stats.key_passes, - "appearances": stats.appearances, - "matches_started": stats.matches_started, - "saves": stats.saves, - "saves_outside_box": stats.saves_outside_box, - "goals_conceded": stats.goals_conceded, - "goals_prevented": stats.goals_prevented, - "high_claims": stats.high_claims, - "penalty_conceded": stats.penalty_conceded, - "penalty_faced": stats.penalty_faced, - "total_shots": stats.total_shots, - "shots_on_target": stats.shots_on_target, - "shots_off_target": stats.shots_off_target, - "scoring_frequency": stats.scoring_frequency, - "penalty_miss": stats.penalty_miss, - "headed_goals": stats.headed_goals, - "left_foot_goals": stats.left_foot_goals, - "right_foot_goals": stats.right_foot_goals, - } + return asdict(stats) def _stats_from_dict(d: dict) -> Stats: - return Stats( - goals=d.get("goals", 0), - assists=d.get("assists", 0), - xg=d.get("xg", 0.0), - xa=d.get("xa", 0.0), - minutes=d.get("minutes", 0), - clean_sheets=d.get("clean_sheets", 0), - pk_saved=d.get("pk_saved", 0), - pk_won=d.get("pk_won", 0), - pk_scored=d.get("pk_scored", 0), - pk_taken=d.get("pk_taken", 0), - yellow_cards=d.get("yellow_cards", 0), - red_cards=d.get("red_cards", 0), - yellow_red_cards=d.get("yellow_red_cards", 0), - direct_red_cards=d.get("direct_red_cards", 0), - fouls_committed=d.get("fouls_committed", 0.0), - rating=d.get("rating", 0.0), - big_chances_created=d.get("big_chances_created", 0), - key_passes=d.get("key_passes", 0), - appearances=d.get("appearances", 0), - matches_started=d.get("matches_started", 0), - saves=d.get("saves", 0), - saves_outside_box=d.get("saves_outside_box", 0), - goals_conceded=d.get("goals_conceded", 0), - goals_prevented=d.get("goals_prevented", 0.0), - high_claims=d.get("high_claims", 0), - penalty_conceded=d.get("penalty_conceded", 0), - penalty_faced=d.get("penalty_faced", 0), - total_shots=d.get("total_shots", 0), - shots_on_target=d.get("shots_on_target", 0), - shots_off_target=d.get("shots_off_target", 0), - scoring_frequency=d.get("scoring_frequency", 0.0), - penalty_miss=d.get("penalty_miss", 0), - headed_goals=d.get("headed_goals", 0), - left_foot_goals=d.get("left_foot_goals", 0), - right_foot_goals=d.get("right_foot_goals", 0), - ) + """Build Stats from a stored doc, ignoring keys the model no longer declares.""" + valid = {f.name for f in fields(Stats)} + return Stats(**{k: v for k, v in d.items() if k in valid}) def _bio_doc(player: PlayerDTO) -> dict: diff --git a/backend/app/modes/fetch_runner.py b/backend/app/modes/fetch_runner.py index ecd7923..6ade321 100644 --- a/backend/app/modes/fetch_runner.py +++ b/backend/app/modes/fetch_runner.py @@ -7,6 +7,7 @@ from app.config import settings from app.domain.competitions import canonical_competition, classify_competition +from app.domain.defensive_stats import apply_defensive_raw from app.domain.models import CompetitionEntry, Stats from app.domain.player_assembler import build_player, merge from app.infrastructure.sofascore_client import SofascoreClient @@ -170,11 +171,13 @@ def _fetch_ss(task_idx: int, comp: str, positions: list[str]) -> None: left_foot_goals=int(row.get("left_foot_goals", 0)), right_foot_goals=int(row.get("right_foot_goals", 0)), ) - position = str(row.get("_position_group") or row.get("position", "MF")) - score = _scoring.calculate(stats, position) raw_stats = row.get("_raw_stats") or {} if not isinstance(raw_stats, dict): raw_stats = {} + apply_defensive_raw(stats, raw_stats) + + position = str(row.get("_position_group") or row.get("position", "MF")) + score = _scoring.calculate(stats, position) canon = canonical_competition(comp) entry = CompetitionEntry( competition=canon, diff --git a/backend/scripts/DB/backfill_defensive_stats.py b/backend/scripts/DB/backfill_defensive_stats.py new file mode 100644 index 0000000..3b79897 --- /dev/null +++ b/backend/scripts/DB/backfill_defensive_stats.py @@ -0,0 +1,107 @@ +"""Backfill the typed defensive Stats fields from each entry's stored raw_stats. + +No re-fetch: the Sofascore defensive columns were always scraped into +competitions[].raw_stats, they were just never promoted into the typed model. Run from +backend/ with Mongo reachable: + + python scripts/DB/backfill_defensive_stats.py # apply + python scripts/DB/backfill_defensive_stats.py --dry-run # report only + +Scores are left untouched: ScoringEngine does not read these fields. + +Stop any running fetch job first: this rewrites each doc's whole competitions array, so a +concurrent fetch writing between the read and the write would be overwritten. +""" + +import os +import sys +from dataclasses import asdict + +from pymongo import MongoClient + +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", ".."))) + +from app.domain.defensive_stats import DEFENSIVE_RAW_MAP, apply_defensive_raw # noqa: E402 +from app.domain.models import Stats # noqa: E402 + +DB_NAME = "football_analytics" + + +def _mongo_uri() -> str: + return os.environ.get("MONGO_URI", "mongodb://localhost:27017/football_analytics") + + +def _stats_from_dict(d: dict) -> Stats: + valid = set(Stats().__dict__) + return Stats(**{k: v for k, v in (d or {}).items() if k in valid}) + + +def _aggregate(entries: list[dict]) -> Stats: + from app.domain.defensive_stats import recompute_rates + + total = Stats() + for entry in entries: + s = _stats_from_dict(entry.get("stats")) + for field in DEFENSIVE_RAW_MAP.values(): + setattr(total, field, getattr(total, field) + getattr(s, field)) + recompute_rates(total) + return total + + +def main() -> None: + dry_run = "--dry-run" in sys.argv + db = MongoClient(_mongo_uri())[DB_NAME] + + scanned = changed = written = entries_touched = no_raw = 0 + + for doc in db.player_stats.find({}): + scanned += 1 + entries = doc.get("competitions") or [] + updated_entries = [] + doc_changed = False + + for entry in entries: + stats = _stats_from_dict(entry.get("stats")) + raw = entry.get("raw_stats") or {} + if not raw: + no_raw += 1 + before = asdict(stats) + apply_defensive_raw(stats, raw) + if asdict(stats) != before: + doc_changed = True + entries_touched += 1 + entry["stats"] = asdict(stats) + updated_entries.append(entry) + + # Write every doc, not only changed ones. A doc left without the new keys is not + # merely stale: Mongo's $gte does not match a missing field, so those players + # disappear from defensive filters, while the stats_view path (which defaults them + # to 0 in Python) still returns them. Same query, two answers. + agg = _stats_from_dict(doc.get("aggregated_stats")) + totals = _aggregate(updated_entries) + for field in list(DEFENSIVE_RAW_MAP.values()) + [ + "tackles_won_pct", + "aerial_duels_won_pct", + ]: + setattr(agg, field, getattr(totals, field)) + + changed += 1 if doc_changed else 0 + if not dry_run: + written += 1 + db.player_stats.update_one( + {"_id": doc["_id"]}, + {"$set": {"competitions": updated_entries, "aggregated_stats": asdict(agg)}}, + ) + + verb = "would change" if dry_run else "changed" + print(f"scanned {scanned} player_stats docs") + print(f"{verb} {changed} docs across {entries_touched} competition entries") + if dry_run: + print(f"would write all {scanned} docs (nothing was written: --dry-run)") + else: + print(f"wrote {written} docs, so none is left without the new keys") + print(f"entries with no raw_stats: {no_raw}") + + +if __name__ == "__main__": + main() diff --git a/backend/tests/domain/test_defensive_stats.py b/backend/tests/domain/test_defensive_stats.py new file mode 100644 index 0000000..c548a00 --- /dev/null +++ b/backend/tests/domain/test_defensive_stats.py @@ -0,0 +1,183 @@ +import pytest + +from app.domain.defensive_stats import DEFENSIVE_RAW_MAP, apply_defensive_raw, recompute_rates +from app.domain.metric_fields import METRIC_FIELDS +from app.domain.models import CompetitionEntry, Score, Stats +from app.domain.player_assembler import aggregate_stats + +DEFENSIVE_COUNTS = [ + "tackles", + "tackles_won", + "interceptions", + "clearances", + "blocks", + "aerial_duels_won", + "aerial_lost", + "ball_recoveries", + "dribbled_past", + "errors_lead_to_goal", + "errors_lead_to_shot", +] +DEFENSIVE_RATES = ["tackles_won_pct", "aerial_duels_won_pct"] + + +def _entry(competition: str, **stats) -> CompetitionEntry: + return CompetitionEntry( + competition=competition, + stats=Stats(**stats), + scores=Score(offensive=0.0, defensive=0.0, tactical=0.0, s_final=0.0), + raw_stats={}, + total_matches=38, + competition_type="club", + ) + + +def test_stats_declares_every_defensive_field_defaulting_to_zero(): + s = Stats() + for field in DEFENSIVE_COUNTS + DEFENSIVE_RATES: + assert getattr(s, field) == 0, f"{field} should default to 0" + + +def test_defensive_fields_are_queryable_metrics(): + for field in DEFENSIVE_COUNTS + DEFENSIVE_RATES: + assert field in METRIC_FIELDS, f"{field} must be sortable/filterable" + + +def test_apply_defensive_raw_maps_sofascore_columns(): + stats = Stats() + apply_defensive_raw( + stats, + { + "tackles": 46.0, + "tacklesWon": 30.0, + "interceptions": 50.0, + "clearances": 164.0, + "outfielderBlocks": 12.0, + "aerialDuelsWon": 60.0, + "aerialLost": 40.0, + "ballRecovery": 88.0, + "dribbledPast": 21.0, + "errorLeadToGoal": 1.0, + "errorLeadToShot": 3.0, + }, + ) + assert stats.tackles == 46 + assert stats.tackles_won == 30 + assert stats.interceptions == 50 + assert stats.clearances == 164 + assert stats.blocks == 12 + assert stats.aerial_duels_won == 60 + assert stats.aerial_lost == 40 + assert stats.ball_recoveries == 88 + assert stats.dribbled_past == 21 + assert stats.errors_lead_to_goal == 1 + assert stats.errors_lead_to_shot == 3 + + +def test_apply_defensive_raw_tolerates_missing_and_null_columns(): + # GK rows and older documents do not carry every column. + stats = Stats() + apply_defensive_raw(stats, {"tackles": None, "interceptions": 7.0}) + assert stats.tackles == 0 + assert stats.interceptions == 7 + assert stats.clearances == 0 + + +def test_recompute_rates_derives_percentages_from_counts(): + stats = Stats(tackles=100, tackles_won=54, aerial_duels_won=60, aerial_lost=40) + recompute_rates(stats) + assert stats.tackles_won_pct == pytest.approx(54.0) + assert stats.aerial_duels_won_pct == pytest.approx(60.0) + + +def test_recompute_rates_is_zero_when_the_denominator_is_zero(): + stats = Stats(tackles=0, tackles_won=0, aerial_duels_won=0, aerial_lost=0) + recompute_rates(stats) + assert stats.tackles_won_pct == 0.0 + assert stats.aerial_duels_won_pct == 0.0 + + +def test_aggregate_sums_defensive_counts_across_competitions(): + agg = aggregate_stats( + [ + _entry("England Premier League", minutes=900, tackles=46, interceptions=50, blocks=4), + _entry("UEFA Champions League", minutes=450, tackles=14, interceptions=10, blocks=2), + ] + ) + assert agg.tackles == 60 + assert agg.interceptions == 60 + assert agg.blocks == 6 + + +def test_aggregate_recomputes_tackle_success_from_summed_counts_not_averaged(): + # 90% over 10 tackles and 50% over 90 tackles average to 70%, but the true + # combined rate is 54/100 = 54%. Averaging percentages is the bug this pins. + agg = aggregate_stats( + [ + _entry("England Premier League", minutes=900, tackles=10, tackles_won=9), + _entry("UEFA Champions League", minutes=900, tackles=90, tackles_won=45), + ] + ) + assert agg.tackles == 100 + assert agg.tackles_won == 54 + assert agg.tackles_won_pct == pytest.approx(54.0) + + +def test_aggregate_recomputes_aerial_success_from_summed_counts(): + agg = aggregate_stats( + [ + _entry("England Premier League", minutes=900, aerial_duels_won=9, aerial_lost=1), + _entry("UEFA Champions League", minutes=900, aerial_duels_won=45, aerial_lost=45), + ] + ) + assert agg.aerial_duels_won == 54 + assert agg.aerial_lost == 46 + assert agg.aerial_duels_won_pct == pytest.approx(54.0) + + +def test_defensive_raw_map_targets_only_real_stats_fields(): + valid = {f for f in Stats().__dict__} + for raw_key, field in DEFENSIVE_RAW_MAP.items(): + assert field in valid, f"{raw_key} maps to unknown Stats field {field}" + + +def test_defensive_fields_survive_the_mongo_round_trip(): + from app.infrastructure.mongo_repository import _stats_from_dict, _stats_to_dict + + original = Stats(minutes=900, tackles=46, tackles_won=30, interceptions=50, clearances=164) + restored = _stats_from_dict(_stats_to_dict(original)) + assert restored == original + + +def test_stats_from_dict_ignores_unknown_keys_from_older_documents(): + from app.infrastructure.mongo_repository import _stats_from_dict + + restored = _stats_from_dict({"goals": 3, "a_field_we_removed": 99}) + assert restored.goals == 3 + assert restored.tackles == 0 + + +def test_stats_out_exposes_every_stats_field(): + from app.api.modals.player_modals import StatsOut + + missing = set(Stats().__dict__) - set(StatsOut.model_fields) + assert not missing, f"StatsOut silently drops: {sorted(missing)}" + + +def test_frontend_metric_options_mirror_the_backend_allowlist(): + # players.ts claims to mirror METRIC_FIELDS. aerial_lost was typed and allowlisted + # but had no dropdown entry, so it could not be filtered from the UI. + import re + from pathlib import Path + + players_ts = Path(__file__).resolve().parents[2].parent / "frontend/src/api/players.ts" + if not players_ts.is_file(): + pytest.skip("frontend not present (backend-only checkout or container)") + after = players_ts.read_text(encoding="utf-8").split("METRIC_OPTIONS")[1] + block = after[: after.index("\n]")] + options = set(re.findall(r"\{ value: '([a-z_0-9]+)'", block)) + assert options, "could not parse METRIC_OPTIONS; the extractor needs updating" + assert options == set(METRIC_FIELDS), ( + f"only in backend: {sorted(set(METRIC_FIELDS) - options)}; " + f"only in frontend: {sorted(options - set(METRIC_FIELDS))}" + ) diff --git a/docs/superpowers/specs/2026-07-11-defender-representation-design.md b/docs/superpowers/specs/2026-07-11-defender-representation-design.md new file mode 100644 index 0000000..ad86e98 --- /dev/null +++ b/docs/superpowers/specs/2026-07-11-defender-representation-design.md @@ -0,0 +1,154 @@ +# Defender Representation — Design Spec + +**Date:** 2026-07-11 +**Status:** Layer 1 implemented 2026-09-06 on `dev/defender-metrics`. Layer 2 is obsolete. +Layer 3 is still an open proposal. +**Branch:** `dev/defender-metrics` + +## Status update — 2026-09-06 + +**Layer 1 is done.** Thirteen defensive fields are typed on `Stats`, populated from +`raw_stats` on both the live-fetch path and a one-off backfill, aggregated as counts with +the two rates recomputed from the summed counts. They are in `METRIC_FIELDS`, so they are +sortable and filterable, exposed by `StatsOut`, and mirrored in the frontend `Stats` and +`METRIC_OPTIONS`. The backfill updated 1235 of 1256 `player_stats` docs across 1394 +competition entries; no re-fetch was needed, exactly as this spec predicted. + +**Field set — question 1 answered by trimming.** Shipped: `tackles`, `tackles_won`, +`interceptions`, `clearances`, `blocks`, `aerial_duels_won`, `aerial_lost`, +`ball_recoveries`, `dribbled_past`, `errors_lead_to_goal`, `errors_lead_to_shot`, plus +the rates `tackles_won_pct` and `aerial_duels_won_pct`. Dropped for now: `duels_won_pct` +(overlaps the aerial rate) and `pass_accuracy_pct` (a ball-playing metric, not a +defending one). Both remain available in `raw_stats` if wanted later. + +**Layer 2 is obsolete.** It targeted the RAG document builder +(`backend/app/domain/player_document.py`) for embedding quality. RAG was cancelled and +that module no longer exists. The consumer of Layer 1 is now the Rankings filters and the +chatbot agent's `defending` tool, which read `METRIC_FIELDS` directly — no prose needed. + +**Layer 3 is unchanged and still needs sign-off.** Scoring was deliberately left out: +`ScoringEngine` does not read these fields, so `s_final` is untouched and no +`Mathematical_Specification.md` update is required by this change. Questions 3 and 4 +below are still open. + +## Problem + +Defenders are currently evaluated and described through an **attacking lens**. Both +the RAG document builder (`playground/document_builder.py` → `backend/app/domain/player_document.py`) +and the `s_final` scoring see only the typed `Stats` model, which is an attacking + +goalkeeping schema. Its only defensive signal is `clean_sheets` (a team-level proxy) +and `fouls_committed`. + +Concretely, `_outfield_headline()` picks a defender's adjective from *finishing* and +*creativity* traits only; when neither fires it emits the static string +`"defensively solid"` with **no data behind it**. Result: semantic retrieval for +queries like *"solid defender"* ranks players on attacking prose, and star defenders +(whose docs emphasise their goals/assists) are mis-ranked. + +## Data availability — the metrics already exist + +Every defensive stat we want is **already scraped and stored** in each competition +entry's untyped `raw_stats` blob; we simply never promote it into the typed model. +Verified present on real player docs: + +| Concept | `raw_stats` key | +|---|---| +| Tackles / won / win % | `tackles`, `tacklesWon`, `tacklesWonPercentage` | +| Interceptions | `interceptions` | +| Clearances | `clearances` | +| Blocks | `outfielderBlocks`, `blockedShots` | +| Aerial duels won / win % | `aerialDuelsWon`, `aerialDuelsWonPercentage`, `aerialLost` | +| Ground/total duels win % | `groundDuelsWonPercentage`, `totalDuelsWonPercentage` | +| Ball recoveries | `ballRecovery` | +| Dribbled past (negative) | `dribbledPast` | +| Errors (negative) | `errorLeadToGoal`, `errorLeadToShot` | +| Passing quality (ball-players) | `accuratePassesPercentage`, `accurateLongBallsPercentage` | + +No new scraping is required. + +## Layer 1 — Promote defensive fields into `Stats` + +Add these fields to `Stats` in `backend/app/domain/models.py` and populate them in +`player_assembler` from `raw_stats` (defaulting to 0 when absent — GK/older rows): + +- Counts: `tackles`, `tackles_won`, `interceptions`, `clearances`, `blocks` + (`outfielderBlocks`), `aerial_duels_won`, `aerial_lost`, `ball_recoveries`, + `dribbled_past`, `errors_lead_to_goal`, `errors_lead_to_shot` +- Percentages: `tackles_won_pct`, `aerial_duels_won_pct`, `duels_won_pct` + (`totalDuelsWonPercentage`), `pass_accuracy_pct` + +**Aggregation rule (important):** percentages must be **recomputed from summed +counts** across competitions (e.g. `tackles_won_pct = Σtackles_won / Σtackles`), NOT +averaged — averaging percentages across competitions with different volumes is wrong. +Where a raw count for the denominator is unavailable, keep the season percentage as-is +and document the limitation. + +Mirror the new fields into `StatsOut` (API) and `Stats` (frontend `players.ts`) and add +the useful ones to `METRIC_OPTIONS` / `METRIC_FIELDS` so they become sortable/filterable +(this also makes them answerable on the structured RAG path — e.g. "best tacklers"). + +## Layer 2 — Defender-aware document (OBSOLETE — RAG cancelled) + +Add `_defender_headline(player)` and a defender branch in the summary. Priority order +matches the analyst's mental model: **clean sheets → tackle success → aerial dominance +→ interceptions/clearances → then goals/assists.** + +**Clean sheets — a representation gap, not a data gap.** `clean_sheets` is already +captured (Sofascore `cleanSheet` → typed `Stats`, aggregated, and scored `×4` for DF), +but the document builder currently mentions it **only in the per-competition stat line** — +the outfield *summary prose* and the *aggregated per-90 block* omit it entirely. So the +embedding barely sees it. Surfacing it is part of this layer: clean sheets must appear in +the defender **headline** and in the aggregated defensive line, not just per-competition. + +Proposed headline adjective (first match wins): +- `"commanding"` — high `aerial_duels_won_pct` (≥ 60%) with meaningful clean sheets +- `"combative ball-winner"` — high `tackles_won_pct` (≥ 65%) and interceptions per 90 +- `"ball-playing"` — high `pass_accuracy_pct` (≥ 88%) / accurate long balls +- `"dependable"` — otherwise (replaces the empty `"defensively solid"`) + +Add a **defensive stat line** to the outfield document (populated for all outfielders, +but it is the *lead* for DF): +``` +Defensive: 2.3 tackles/90 (71% won), 1.8 interceptions/90, 3.1 clearances/90, + 68% aerial duels won, 0.4 blocks/90, 0 errors leading to a goal. +``` +Attacking output stays, but for DF it moves **after** the defensive line. + +## Layer 3 — Scoring (PROPOSAL — still needs sign-off, not implemented) + +Optional; only if we choose "docs + scoring". Fold defensive output into `s_final` for +`position == "DF"` (and partially MF). This changes app-wide ranking and requires a +`Mathematical_Specification.md` update. **All weights below are placeholders for review, +not final:** + +| Metric (per 90) | Proposed weight | +|---|---| +| Tackle won | +0.6 | +| Interception | +0.6 | +| Clearance | +0.2 | +| Block | +0.5 | +| Aerial duel won | +0.4 | +| Ball recovery | +0.1 | +| `errorLeadToGoal` | −3.0 | +| `errorLeadToShot` | −1.0 | +| Dribbled past | −0.3 | + +Quality percentages (`tackles_won_pct`, `aerial_duels_won_pct`) act as **multipliers** +on the corresponding volume term rather than standalone additive points, so high-volume +low-quality defenders aren't overrated. Normalisation and `playing_time_factor` / +`starter_bonus` stay as in the current `s_final`. + +## Out of scope + +- No new scraping or data sources. +- No change to GK scoring/representation. +- No re-fetch required — a one-off backfill re-reads existing `raw_stats` into the new + typed fields (a small migration script over `player_stats`). + +## Open questions for review + +1. Field set — is the Layer-1 list right, or trim/extend (e.g. drop `duels_won_pct`)? +2. Headline thresholds (60% aerial, 65% tackle, 88% pass) — reasonable, or tune? +3. Scoring: in scope now, or docs+retrieval only and defer scoring to a follow-up? +4. If scoring is in scope, do the proposed weights need a data-distribution check first + (data-analyst) before we commit numbers to the math spec? diff --git a/frontend/src/api/players.ts b/frontend/src/api/players.ts index a0a6b53..72febc0 100644 --- a/frontend/src/api/players.ts +++ b/frontend/src/api/players.ts @@ -13,6 +13,11 @@ export interface Stats { total_shots: number; shots_on_target: number; shots_off_target: number scoring_frequency: number; penalty_miss: number headed_goals: number; left_foot_goals: number; right_foot_goals: number + tackles: number; tackles_won: number; tackles_won_pct: number + interceptions: number; clearances: number; blocks: number + aerial_duels_won: number; aerial_lost: number; aerial_duels_won_pct: number + ball_recoveries: number; dribbled_past: number + errors_lead_to_goal: number; errors_lead_to_shot: number } export interface Score { @@ -85,6 +90,19 @@ export const METRIC_OPTIONS: { value: string; label: string }[] = [ { value: 'headed_goals', label: 'Headed goals' }, { value: 'left_foot_goals', label: 'Left-foot goals' }, { value: 'right_foot_goals', label: 'Right-foot goals' }, + { value: 'tackles', label: 'Tackles' }, + { value: 'tackles_won', label: 'Tackles won' }, + { value: 'tackles_won_pct', label: 'Tackle success %' }, + { value: 'interceptions', label: 'Interceptions' }, + { value: 'clearances', label: 'Clearances' }, + { value: 'blocks', label: 'Blocks' }, + { value: 'aerial_duels_won', label: 'Aerial duels won' }, + { value: 'aerial_lost', label: 'Aerial duels lost' }, + { value: 'aerial_duels_won_pct', label: 'Aerial duel success %' }, + { value: 'ball_recoveries', label: 'Ball recoveries' }, + { value: 'dribbled_past', label: 'Dribbled past' }, + { value: 'errors_lead_to_goal', label: 'Errors leading to a goal' }, + { value: 'errors_lead_to_shot', label: 'Errors leading to a shot' }, { value: 'offensive', label: 'Offensive score' }, { value: 'defensive', label: 'Defensive score' }, { value: 'tactical', label: 'Tactical score' },