diff --git a/bin/evaluate b/bin/evaluate new file mode 100755 index 00000000..9bad3559 --- /dev/null +++ b/bin/evaluate @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +"""Entry point for the ragas evaluation; see src/evaluation/evaluator.py.""" + +from evaluation.evaluator import main + +if __name__ == "__main__": + main() diff --git a/src/evaluation/README.md b/src/evaluation/README.md index 36fded4b..481b1c36 100644 --- a/src/evaluation/README.md +++ b/src/evaluation/README.md @@ -1,46 +1,76 @@ # RAGAS Evaluation Toolkit -This folder contains utility scripts used to benchmark Reactome RAG pipelines with Ragas. +Scores the answers the chatbot actually gives. -- `test_generator.py` synthesizes question/answer test sets from the example corpora. It uses the Ragas `TestsetGenerator` to create LangChain-based evaluation datasets. -- `evaluator.py` runs either the basic or advanced Reactome RAG chain over a test set and scores the outputs with Ragas metrics (answer relevancy, context utilization, faithfulness, context recall), saving both responses and evaluation reports. +- `evaluator.py` (run it as `./bin/evaluate`) asks the **shipping RAG chain** a set + of questions and scores the answers with ragas: faithfulness, answer relevancy, + context utilization, and context recall when reference answers are supplied. +- `test_generator.py` synthesizes question/answer sets from the example corpora. + **It targets the ragas 0.1 API and does not run against the pinned 0.2** — see + the TODO in the file. `tests/golden/questions.txt` is what the evaluator uses by + default and needs no generation. -## Requirements +## What changed, and why the old flags are gone + +`--rag_type basic|advanced` and `--testset_dir` no longer exist. + +The evaluator used to build its own retriever — `SelfQueryRetriever` + +`EnsembleRetriever` + `MergerRetriever`, over the `summations` collection alone, +with `k=7` and weights `[0.2, 0.8]` that appear nowhere in the product. The +retriever rewrite then removed `SelfQueryRetriever` from the pipeline, so the +evaluator was measuring a configuration that existed nowhere. `basic` and +`advanced` were two shapes of that private stack, not two shapes of the product. -- Python 3.12 (project default) with Poetry environment -- `ragas` (see poetry.lock for the pinned version) -- OpenAI access: `OPENAI_API_KEY` (and optional Azure configuration if required) -- An installed embeddings bundle (`./bin/embeddings_manager install ...`). - `evaluator.py` defaults to whichever bundle is active; override with - `--embeddings-dir`. +It now calls `create_reactome_rag`, the same factory `bin/chat-chainlit.py` uses. +There is only one pipeline to measure, so there is no `--rag_type`. -Run `poetry install` to set up dependencies, then activate the virtual environment via `poetry shell` or use `poetry run` for individual commands. +Reference answers, needed only for `context_recall`, come from `--references` +as JSON rather than from spreadsheets: + +```json +{ "What role does TP53 play in apoptosis?": "TP53 induces apoptosis by ..." } +``` + +## Requirements + +- An installed reactome bundle (`./bin/embeddings_manager install ...`) +- `OPENAI_API_KEY` ## Usage -1. **Generate test sets** +```bash +# one model over the golden questions +./bin/evaluate --model gpt-4o-mini + +# two models, same questions, same judge, side by side +./bin/evaluate --model gpt-4o-mini --model gpt-5.6-luna + +# three runs each, so the report can show the noise floor +./bin/evaluate --model gpt-4o-mini --repeat 3 --out report.json +``` - ```bash - poetry run python src/evaluation/test_generator.py \ - --path src/evaluation/example \ - --model gpt-4o-mini \ - --temperature 0.3 \ - --test_size 10 \ - --distributions simple=0.25 reasoning=0.25 multi_context=0.25 conditional=0.25 - ``` +`--out` writes the full report: aggregate scores per run, seconds per question, +and every answer with its per-question scores — so a low score can be looked at +rather than guessed about. - Outputs are stored in a `testsets/` directory of your choosing. +## The judge -2. **Evaluate a RAG configuration** +Scoring is done by a separate model, `gpt-4o` by default, pinned with +`--judge-model`. Two rules the tool enforces or documents: - ```bash - poetry run python src/evaluation/evaluator.py \ - --testset_dir \ - --rag_type advanced \ - --model gpt-4o-mini - ``` +- **The judge may not be a model under test.** The tool refuses. A model grading + its own answers is not a measurement. +- **The judge must not change between runs being compared.** A moving judge makes + two runs incomparable, which is the failure this tool exists to avoid. - Responses and metric reports are written to `response//` and `evals//` inside the testset directory. +The judge's embedding model is only used to compare a question against an answer; +it is unrelated to the vectors in the bundle, and it is pointed at +`api.openai.com` explicitly rather than following `OPENAI_BASE_URL`, which on the +Plant Reactome host points at a self-hosted endpoint that does not serve it. +`JUDGE_BASE_URL` overrides. -Adjust the paths, model names, and distribution weights as needed for local experimentation. +## Reading the output +A single run has no noise floor: retrieval is not deterministic (Chroma's ANN +search varies run to run), so a difference between two single runs cannot be told +apart from variance. Use `--repeat 3` and compare against the reported spread. diff --git a/src/evaluation/evaluator.py b/src/evaluation/evaluator.py index b4ecf357..43f3e74d 100644 --- a/src/evaluation/evaluator.py +++ b/src/evaluation/evaluator.py @@ -1,229 +1,342 @@ +"""Score the answers the chatbot actually gives, with ragas. + +This measures **the shipping pipeline**. That is the whole point of the file and +was not true of it until now: it used to build its own `SelfQueryRetriever` + +`EnsembleRetriever` + `MergerRetriever` over the `summations` collection alone, +with `k=7` and weights `[0.2, 0.8]` that appear nowhere in the product. After the +retriever rewrite removed `SelfQueryRetriever` from the pipeline, it measured a +configuration that no longer existed anywhere -- producing numbers that looked +like an answer and were not. + +It now calls `create_reactome_rag`, the same factory `bin/chat-chainlit.py` +reaches, so a change to retrieval cannot alter the product without altering the +measurement. + + # one model over the golden questions + ./bin/evaluate --model gpt-4o-mini + + # two models, same questions, same judge, side by side + ./bin/evaluate --model gpt-4o-mini --model gpt-5.6-luna + + # how much of a difference is just noise? + ./bin/evaluate --model gpt-4o-mini --repeat 3 + +Requires an installed reactome bundle and OPENAI_API_KEY. +""" + import argparse +import json +import math import os +import statistics +import sys +import time from pathlib import Path from typing import Any -import pandas as pd -from datasets import Dataset -from langchain.retrievers import EnsembleRetriever -from langchain.retrievers.merger_retriever import MergerRetriever -from langchain.retrievers.self_query.base import SelfQueryRetriever -from langchain_chroma.vectorstores import Chroma -from langchain_community.document_loaders.csv_loader import CSVLoader -from langchain_community.retrievers import BM25Retriever -from langchain_core.retrievers import BaseRetriever -from langchain_core.runnables import Runnable -from langchain_openai import ChatOpenAI, OpenAIEmbeddings -from ragas import evaluate +import nltk +from dotenv import load_dotenv +from langchain_core.language_models.chat_models import BaseChatModel +from ragas import EvaluationDataset, SingleTurnSample, evaluate +from ragas.embeddings import LangchainEmbeddingsWrapper +from ragas.llms import LangchainLLMWrapper from ragas.metrics import ( ContextUtilization, - answer_relevancy, - context_recall, - faithfulness, + Faithfulness, + LLMContextRecall, + ResponseRelevancy, ) -from retrievers.rag_chain import create_rag_chain -from retrievers.reactome.metadata_info import ( - reactome_descriptions_info, - reactome_field_info, -) -from retrievers.reactome.prompt import reactome_qa_prompt +from agent.graph import resolve_embedding_model, resolve_temperature +from agent.models import get_embedding, get_llm +from retrievers.reactome.rag import create_reactome_rag from util.embedding_environment import EmbeddingEnvironment -context_utilization = ContextUtilization() +REPO_ROOT = Path(__file__).parent.parent.parent +DEFAULT_QUESTIONS = REPO_ROOT / "tests" / "golden" / "questions.txt" + +# The judge is pinned and is NOT the model under test. Scoring gpt-5.6-luna with +# gpt-5.6-luna would ask a model to grade its own homework; and a judge that moves +# between two runs makes the two runs incomparable, which is the failure this +# whole file exists to avoid. +DEFAULT_JUDGE_MODEL = "gpt-4o" + +# Used only to compare a question against an answer for answer_relevancy. +JUDGE_EMBEDDING_MODEL = "text-embedding-3-large" + + +def read_questions(path: Path) -> list[str]: + lines = path.read_text().splitlines() + return [ln.strip() for ln in lines if ln.strip() and not ln.startswith("#")] + + +def read_references(path: Path) -> dict[str, str]: + """Optional question -> reference answer map, as JSON. + + Only `context_recall` needs one. Without it that metric is dropped rather + than scored against nothing. + """ + data: dict[str, str] = json.loads(path.read_text()) + return data + + +def answer_questions( + chain: Any, questions: list[str] +) -> tuple[list[str], list[list[str]], float]: + """Ask the chain each question, keeping the answer and the retrieved context.""" + answers: list[str] = [] + contexts: list[list[str]] = [] + started = time.monotonic() + for i, question in enumerate(questions, start=1): + print(f" [{i}/{len(questions)}] {question[:70]}", file=sys.stderr) + response = chain.invoke({"input": question, "chat_history": []}) + answers.append(response["answer"]) + contexts.append([doc.page_content for doc in response["context"]]) + return answers, contexts, time.monotonic() - started + + +def build_chain(model: str, embeddings_dir: Path) -> Any: + """The chain under test, built exactly as the application builds it.""" + llm: BaseChatModel = get_llm( + "openai", + model, + request_timeout=360.0, + temperature=resolve_temperature(model), + ) + embedding = get_embedding("openai", resolve_embedding_model()) + return create_reactome_rag(llm, embedding, embeddings_dir) + + +def score( + questions: list[str], + answers: list[str], + contexts: list[list[str]], + references: dict[str, str], + judge: str, +) -> tuple[dict[str, float], list[dict[str, float]]]: + judge_llm = LangchainLLMWrapper( + get_llm("openai", judge, temperature=resolve_temperature(judge)) + ) + # answer_relevancy embeds the question and the answer to compare them. This + # is the JUDGE's embedding and has nothing to do with the vectors in the + # bundle, so it does not go through resolve_embedding_model -- but it must + # not silently follow OPENAI_BASE_URL either. On the Plant Reactome host that + # points at a self-hosted bge-m3 endpoint, and asking that for + # text-embedding-3-large is a 404 in the middle of a run. api.openai.com is + # named explicitly; JUDGE_BASE_URL overrides it. + judge_embeddings = LangchainEmbeddingsWrapper( + get_embedding( + "openai", + JUDGE_EMBEDDING_MODEL, + base_url=os.getenv("JUDGE_BASE_URL", "https://api.openai.com/v1"), + ) + ) + + samples = [ + SingleTurnSample( + user_input=q, + response=a, + retrieved_contexts=c, + reference=references.get(q), + ) + for q, a, c in zip(questions, answers, contexts, strict=True) + ] + + metrics: list[Any] = [Faithfulness(), ResponseRelevancy(), ContextUtilization()] + if all(s.reference for s in samples): + metrics.append(LLMContextRecall()) + else: + print( + " (no reference answers, so context_recall is skipped rather than " + "scored against nothing -- pass --references to include it)", + file=sys.stderr, + ) + + result = evaluate( + dataset=EvaluationDataset(samples=samples), + metrics=metrics, + llm=judge_llm, + embeddings=judge_embeddings, + ) + # result.scores is a public field holding one dict of metric -> score per + # sample. The aggregate used to come from result._repr_dict, which is private + # and would break on a ragas upgrade without warning -- in a file whose whole + # job is to stay trustworthy across upgrades. + per_question: list[dict[str, float]] = [dict(s) for s in result.scores] + + # A metric that fails on one sample comes back NaN, and NaN propagates + # through fmean -- so one bad question would turn the whole aggregate into + # NaN, which prints as "nan" and looks like a broken tool rather than a + # partial result. Non-finite scores are dropped and the drop is reported, + # because silently averaging over fewer questions than were asked is the kind + # of quiet difference this file is supposed to catch, not commit. + aggregate: dict[str, float] = {} + for metric in per_question[0]: + values = [ + s[metric] + for s in per_question + if s.get(metric) is not None and math.isfinite(s[metric]) + ] + dropped = len(per_question) - len(values) + if dropped: + print( + f" {metric}: {dropped} of {len(per_question)} questions could " + "not be scored and are excluded from the mean", + file=sys.stderr, + ) + aggregate[metric] = statistics.fmean(values) if values else float("nan") + return aggregate, per_question -def parse_arguments() -> argparse.Namespace: - """Parse command line arguments for the script.""" +def main() -> None: + load_dotenv() + try: + nltk.data.find("tokenizers/punkt_tab") + except LookupError: + nltk.download("punkt_tab", quiet=True) + parser = argparse.ArgumentParser( - description="Load a directory of testsets and evaluate answers generated by a language model." + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter ) parser.add_argument( - "--testset_dir", - type=str, - required=True, - help="Path to the directory containing testset Excel (.xlsx) files", + "--model", + action="append", + dest="models", + help="Model under test. Repeat to compare several against one judge.", ) parser.add_argument( - "--embeddings-dir", + "--judge-model", + default=DEFAULT_JUDGE_MODEL, + help=f"Model that scores the answers (default: {DEFAULT_JUDGE_MODEL}).", + ) + parser.add_argument("--questions", type=Path, default=DEFAULT_QUESTIONS) + parser.add_argument( + "--references", type=Path, - default=EmbeddingEnvironment.get_dir("reactome"), - help=( - "Reactome embeddings bundle to evaluate against. Defaults to the " - "installed one (see ./bin/embeddings_manager which)." - ), + help="JSON {question: reference answer}; enables context_recall.", ) parser.add_argument( - "--model", - type=str, - default="gpt-4o-mini", - help="Language model to use for evaluation", + "--repeat", + type=int, + default=1, + help="Run each model this many times, and report the spread as the noise " + "floor. A difference smaller than it is not a result.", ) + parser.add_argument("--limit", type=int, help="Use only the first N questions.") parser.add_argument( - "--rag_type", - choices=["basic", "advanced"], - required=True, - help="Type of RAG system to use for evaluation", + "--embeddings-dir", type=Path, default=EmbeddingEnvironment.get_dir("reactome") ) - return parser.parse_args() + parser.add_argument("--out", type=Path, help="Write the full report as JSON.") + args = parser.parse_args() + models: list[str] = args.models or ["gpt-4o-mini"] + if args.judge_model in models: + raise SystemExit( + f"The judge ({args.judge_model}) is also under test. A model grading " + "its own answers is not a measurement. Pass a different --judge-model." + ) + embeddings_dir: Path | None = args.embeddings_dir + if embeddings_dir is None: + raise SystemExit( + "No reactome embeddings installed. Run " + "./bin/embeddings_manager install , or pass " + "--embeddings-dir." + ) -def load_dataset(testset_path: str) -> list[dict[str, Any]]: - """Load the dataset from an Excel (.xlsx) file.""" - # pandas types record keys as Hashable; they are column names. - try: - df = pd.read_excel(testset_path) - records: list[dict[str, Any]] = df.to_dict(orient="records") # type: ignore[assignment] - return records - except FileNotFoundError as e: - raise FileNotFoundError(f"The file {testset_path} does not exist.") from e - except ValueError as e: - raise ValueError(f"Error reading the Excel file: {e}") from e - - -def initialize_rag_chain_with_memory( - embeddings_dir: Path, model_name: str, rag_type: str -) -> Runnable: - """Initialize the RAGChainWithMemory system. - - `embeddings_dir` is the bundle root, e.g. - embeddings/openai/text-embedding-3-large/reactome/Release95 . Both the BM25 - source CSV and the Chroma collection are derived from it; they used to be - absolute paths into a developer's home directory, so this script could not - run anywhere else. - """ - llm = ChatOpenAI(temperature=0.0, verbose=True, model=model_name) - retriever_list: list[BaseRetriever] = [] - - loader = CSVLoader(str(embeddings_dir / "csv_files" / "summations.csv")) - data = loader.load() - bm25_retriever = BM25Retriever.from_documents(data) - bm25_retriever.k = 7 - - # Set up vectorstore SelfQuery retriever - embedding = OpenAIEmbeddings(model="text-embedding-3-large") - vectordb = Chroma( - persist_directory=str(embeddings_dir / "summations"), - embedding_function=embedding, - ) - - vectordb_retriever = vectordb.as_retriever(search_kwargs={"k": 7}) + questions = read_questions(args.questions) + if args.limit: + questions = questions[: args.limit] + references = read_references(args.references) if args.references else {} - selfq_retriever = SelfQueryRetriever.from_llm( - llm=llm, - vectorstore=vectordb, - document_contents=reactome_descriptions_info["summations"], - metadata_field_info=reactome_field_info["summations"], - search_kwargs={"k": 7}, - ) - rrf_retriever = EnsembleRetriever( - retrievers=[bm25_retriever, selfq_retriever], weights=[0.2, 0.8] + print( + f"{len(questions)} questions x {len(models)} model(s) x {args.repeat} run(s), " + f"judged by {args.judge_model}\nbundle: {embeddings_dir}\n", + file=sys.stderr, ) - if rag_type == "basic": - retriever_list.append(vectordb_retriever) - elif rag_type == "advanced": - retriever_list.append(rrf_retriever) - reactome_retriever = MergerRetriever(retrievers=retriever_list) + report: dict[str, Any] = { + "judge_model": args.judge_model, + "bundle": str(embeddings_dir), + "questions": len(questions), + "repeat": args.repeat, + "models": {}, + } - return create_rag_chain( - retriever=reactome_retriever, - llm=llm, - qa_prompt=reactome_qa_prompt, - ) + for model in models: + runs: list[dict[str, float]] = [] + seconds: list[float] = [] + transcripts: list[list[dict[str, Any]]] = [] + chain = build_chain(model, embeddings_dir) + for run in range(1, args.repeat + 1): + print(f" {model} run {run}/{args.repeat}", file=sys.stderr) + answers, contexts, elapsed = answer_questions(chain, questions) + seconds.append(elapsed / len(questions)) + aggregate, per_question = score( + questions, answers, contexts, references, args.judge_model + ) + runs.append(aggregate) + # The answers themselves, not only the scores. The version this + # replaced wrote responses to a spreadsheet; dropping that would have + # left a low faithfulness score with nothing to look at to find out + # why. + transcripts.append( + [ + { + "question": q, + "answer": a, + "documents_retrieved": len(c), + "scores": s, + } + for q, a, c, s in zip( + questions, answers, contexts, per_question, strict=True + ) + ] + ) + report["models"][model] = { + "runs": runs, + "seconds_per_question": seconds, + "transcripts": transcripts, + } + print_report(report) + if args.out: + args.out.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n") + print(f"\nWrote {args.out}", file=sys.stderr) -def process_testset( - testset_path: str, - qa_system: Runnable, - response_dir: str, - eval_dir: str, - model_name: str, - rag_type: str, -) -> None: - """Process a single testset file.""" - testset = load_dataset(testset_path) - questions = [item["question"] for item in testset] - ground_truths = [item["ground_truth"] for item in testset] - - answers = [] - contexts = [] - - for question in questions: - response = qa_system.invoke({"input": question}) - answers.append(response["answer"]) - contexts.append([context.page_content for context in response["context"]]) - - rag_response_dir = os.path.join(response_dir, rag_type) - rag_eval_dir = os.path.join(eval_dir, rag_type) - os.makedirs(rag_response_dir, exist_ok=True) - os.makedirs(rag_eval_dir, exist_ok=True) - - # Save responses to an Excel file - data = { - "question": questions, - "answer": answers, - "contexts": contexts, - "ground_truth": ground_truths, - } - df_ans = pd.DataFrame(data) - response_filename = os.path.join( - rag_response_dir, - f"{os.path.splitext(os.path.basename(testset_path))[0]}_{model_name}_responses_{rag_type}.xlsx", - ) - df_ans.to_excel(response_filename, index=False) - print(f"Responses saved to {response_filename}") - # Evaluate the dataset - dataset = Dataset.from_dict(data) - result = evaluate( - llm=ChatOpenAI(temperature=0.0, verbose=True, model="gpt-4o"), - dataset=dataset, - metrics=[answer_relevancy, context_utilization, faithfulness, context_recall], - ) +def print_report(report: dict[str, Any]) -> None: + models: dict[str, Any] = report["models"] + metric_names = sorted({m for v in models.values() for m in v["runs"][0]}) - # Save evaluation results to an Excel file - evaluation_filename = os.path.join( - rag_eval_dir, - f"{os.path.splitext(os.path.basename(testset_path))[0]}_{model_name}_evaluation_{rag_type}.xlsx", + print( + f"\n{report['questions']} questions, judged by {report['judge_model']}, " + f"{report['repeat']} run(s) each\n" ) - df_eval = result.to_pandas() - df_eval.to_excel(evaluation_filename, index=False) - print(f"Evaluation results saved to {evaluation_filename}") - + header = f" {'model':<18}" + "".join(f"{m[:16]:>18}" for m in metric_names) + print(header + f"{'s/question':>13}") + print(" " + "-" * (len(header) + 11)) + for model, data in models.items(): + cells = "" + for metric in metric_names: + values = [r[metric] for r in data["runs"]] + mean = statistics.fmean(values) + spread = (max(values) - min(values)) if len(values) > 1 else 0.0 + cells += ( + f"{mean:>13.3f}±{spread:.3f}" if len(values) > 1 else f"{mean:>18.3f}" + ) + secs = statistics.fmean(data["seconds_per_question"]) + print(f" {model:<18}{cells}{secs:>13.1f}") -def main() -> None: - args = parse_arguments() - model_name = args.model - rag_type = args.rag_type - response_dir = os.path.join(args.testset_dir, "response") - eval_dir = os.path.join(args.testset_dir, "evals") - os.makedirs(response_dir, exist_ok=True) - os.makedirs(eval_dir, exist_ok=True) - - # Initialize RAG Chain - embeddings_dir: Path | None = args.embeddings_dir - if embeddings_dir is None: - raise SystemExit( - "No reactome embeddings installed and --embeddings-dir not given. " - "Install one with ./bin/embeddings_manager install ." + if report["repeat"] > 1: + print( + "\n The ± is the observed spread across runs -- the noise floor. A\n" + " difference between models smaller than it is not a result." + ) + else: + print( + "\n Single run, so there is no noise floor here and no way to tell a\n" + " real difference from run-to-run variance. Use --repeat 3 to compare." ) - qa_system = initialize_rag_chain_with_memory(embeddings_dir, model_name, rag_type) - - # Iterate over all .xlsx files in the directory - for filename in os.listdir(args.testset_dir): - print(f"Found file: {filename}") - if filename.endswith(".xlsx"): - testset_path = os.path.join(args.testset_dir, filename) - print(f"Processing testset: {testset_path}") - process_testset( - testset_path, - qa_system, - response_dir, - eval_dir, - model_name, - rag_type, - ) if __name__ == "__main__": diff --git a/tests/evaluation/test_evaluator.py b/tests/evaluation/test_evaluator.py new file mode 100644 index 00000000..aa0e2d27 --- /dev/null +++ b/tests/evaluation/test_evaluator.py @@ -0,0 +1,51 @@ +"""The parts of the evaluator that can be checked without spending an API call. + +The scoring itself needs the network and a bundle; what is pinned here is the +input handling and the guards -- including the one that stops a model from +grading its own answers, which would silently produce numbers rather than fail. +""" + +from pathlib import Path + +import pytest + +pytest.importorskip("ragas", reason="evaluation stack not installed") + +from evaluation.evaluator import ( # noqa: E402 + DEFAULT_JUDGE_MODEL, + DEFAULT_QUESTIONS, + read_questions, + read_references, +) + + +def test_the_default_question_set_is_the_committed_one() -> None: + """The evaluator and bin/retrieval_baseline must ask the same questions. + + Two measurement tools disagreeing about the question set would make their + results incomparable for no reason. + """ + assert DEFAULT_QUESTIONS.exists(), DEFAULT_QUESTIONS + assert DEFAULT_QUESTIONS.name == "questions.txt" + assert len(read_questions(DEFAULT_QUESTIONS)) >= 20 + + +def test_comments_and_blank_lines_are_not_questions(tmp_path: Path) -> None: + path = tmp_path / "q.txt" + path.write_text("# a heading\n\nWhat is TP53?\n\n#another\n Which complexes? \n") + assert read_questions(path) == ["What is TP53?", "Which complexes?"] + + +def test_references_are_read_as_a_question_to_answer_map(tmp_path: Path) -> None: + path = tmp_path / "refs.json" + path.write_text('{"What is TP53?": "A tumour suppressor."}') + assert read_references(path) == {"What is TP53?": "A tumour suppressor."} + + +def test_the_default_judge_is_not_a_model_this_repo_answers_with() -> None: + """The judge must be pinned and distinct; see the guard in main(). + + gpt-4o-mini is the default answering model, so the default judge must not be + it -- otherwise the out-of-the-box invocation grades its own homework. + """ + assert DEFAULT_JUDGE_MODEL != "gpt-4o-mini"