From cf9ad6abee68f2ff49fecf0a6e92369ea30eb9dc Mon Sep 17 00:00:00 2001 From: Adam Wright Date: Wed, 9 Sep 2026 16:41:21 +0000 Subject: [PATCH] Point the evaluator at the pipeline that ships src/evaluation/evaluator.py runs the right ragas metrics and measured the wrong thing. It built its own retriever -- SelfQueryRetriever + EnsembleRetriever + MergerRetriever, over the `summations` collection alone, with k=7 and weights [0.2, 0.8] that appear nowhere in the product. The retriever rewrite then removed SelfQueryRetriever from the pipeline, so it measured a configuration that existed nowhere at all. Running it would have produced numbers that looked like an answer. It now calls create_reactome_rag, the same factory bin/chat-chainlit.py reaches: four collections, the real budget, the real fusion. There is one pipeline, so --rag_type basic|advanced is gone -- those were two shapes of the private stack, not of the product. The embedding comes from resolve_embedding_model() rather than a hardcoded text-embedding-3-large, so Plant Reactome is not broken by it. This is spec 002's P1 (FR-001..FR-004). It unblocks the gpt-5.6-luna decision and the four retrieval changes from 2026-09-04 that shipped unevaluated. ./bin/evaluate --model gpt-4o-mini --model gpt-5.6-luna --repeat 3 The model under test is an argument and can be repeated; the judge is pinned separately, defaults to gpt-4o, and the tool refuses to run when the judge is also under test, because a model grading its own answers is not a measurement. --repeat reports the spread across runs as the noise floor: retrieval is not deterministic, so a single run cannot distinguish a real difference from Chroma's ANN variance, and the report says so rather than letting the reader assume otherwise. Five things an adversarial review of the rewrite caught, all fixed before this landed: - the aggregate came from result._repr_dict, a private ragas attribute, in a file whose job is to stay trustworthy across upgrades. Now result.scores, which is public. - a metric that fails on one sample returns NaN, and NaN propagates through fmean -- one bad question would have turned the whole aggregate into "nan". Non-finite scores are dropped and the drop is reported. - the judge's embedding was a bare "text-embedding-3-large" with no base_url, so it would follow OPENAI_BASE_URL. On the Plant Reactome host that points at a self-hosted bge-m3 endpoint: a 404 mid-run. api.openai.com is now named explicitly, with JUDGE_BASE_URL to override. - only scores were kept, not answers. The version this replaces wrote responses to a spreadsheet; dropping that would have left a low faithfulness score with nothing to look at. --out now carries every answer with its per-question scores. - the README still documented --testset_dir and --rag_type. Rewritten, including why those flags are gone. Verified by running it: 2 and 3 question sets against the Release95 bundle, real scores, and the judge guard refusing gpt-4o vs gpt-4o. --- bin/evaluate | 7 + src/evaluation/README.md | 90 ++++-- src/evaluation/evaluator.py | 491 ++++++++++++++++++----------- tests/evaluation/test_evaluator.py | 51 +++ 4 files changed, 420 insertions(+), 219 deletions(-) create mode 100755 bin/evaluate create mode 100644 tests/evaluation/test_evaluator.py diff --git a/bin/evaluate b/bin/evaluate new file mode 100755 index 00000000..9bad3559 --- /dev/null +++ b/bin/evaluate @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +"""Entry point for the ragas evaluation; see src/evaluation/evaluator.py.""" + +from evaluation.evaluator import main + +if __name__ == "__main__": + main() diff --git a/src/evaluation/README.md b/src/evaluation/README.md index 36fded4b..481b1c36 100644 --- a/src/evaluation/README.md +++ b/src/evaluation/README.md @@ -1,46 +1,76 @@ # RAGAS Evaluation Toolkit -This folder contains utility scripts used to benchmark Reactome RAG pipelines with Ragas. +Scores the answers the chatbot actually gives. -- `test_generator.py` synthesizes question/answer test sets from the example corpora. It uses the Ragas `TestsetGenerator` to create LangChain-based evaluation datasets. -- `evaluator.py` runs either the basic or advanced Reactome RAG chain over a test set and scores the outputs with Ragas metrics (answer relevancy, context utilization, faithfulness, context recall), saving both responses and evaluation reports. +- `evaluator.py` (run it as `./bin/evaluate`) asks the **shipping RAG chain** a set + of questions and scores the answers with ragas: faithfulness, answer relevancy, + context utilization, and context recall when reference answers are supplied. +- `test_generator.py` synthesizes question/answer sets from the example corpora. + **It targets the ragas 0.1 API and does not run against the pinned 0.2** — see + the TODO in the file. `tests/golden/questions.txt` is what the evaluator uses by + default and needs no generation. -## Requirements +## What changed, and why the old flags are gone + +`--rag_type basic|advanced` and `--testset_dir` no longer exist. + +The evaluator used to build its own retriever — `SelfQueryRetriever` + +`EnsembleRetriever` + `MergerRetriever`, over the `summations` collection alone, +with `k=7` and weights `[0.2, 0.8]` that appear nowhere in the product. The +retriever rewrite then removed `SelfQueryRetriever` from the pipeline, so the +evaluator was measuring a configuration that existed nowhere. `basic` and +`advanced` were two shapes of that private stack, not two shapes of the product. -- Python 3.12 (project default) with Poetry environment -- `ragas` (see poetry.lock for the pinned version) -- OpenAI access: `OPENAI_API_KEY` (and optional Azure configuration if required) -- An installed embeddings bundle (`./bin/embeddings_manager install ...`). - `evaluator.py` defaults to whichever bundle is active; override with - `--embeddings-dir`. +It now calls `create_reactome_rag`, the same factory `bin/chat-chainlit.py` uses. +There is only one pipeline to measure, so there is no `--rag_type`. -Run `poetry install` to set up dependencies, then activate the virtual environment via `poetry shell` or use `poetry run` for individual commands. +Reference answers, needed only for `context_recall`, come from `--references` +as JSON rather than from spreadsheets: + +```json +{ "What role does TP53 play in apoptosis?": "TP53 induces apoptosis by ..." } +``` + +## Requirements + +- An installed reactome bundle (`./bin/embeddings_manager install ...`) +- `OPENAI_API_KEY` ## Usage -1. **Generate test sets** +```bash +# one model over the golden questions +./bin/evaluate --model gpt-4o-mini + +# two models, same questions, same judge, side by side +./bin/evaluate --model gpt-4o-mini --model gpt-5.6-luna + +# three runs each, so the report can show the noise floor +./bin/evaluate --model gpt-4o-mini --repeat 3 --out report.json +``` - ```bash - poetry run python src/evaluation/test_generator.py \ - --path src/evaluation/example \ - --model gpt-4o-mini \ - --temperature 0.3 \ - --test_size 10 \ - --distributions simple=0.25 reasoning=0.25 multi_context=0.25 conditional=0.25 - ``` +`--out` writes the full report: aggregate scores per run, seconds per question, +and every answer with its per-question scores — so a low score can be looked at +rather than guessed about. - Outputs are stored in a `testsets/` directory of your choosing. +## The judge -2. **Evaluate a RAG configuration** +Scoring is done by a separate model, `gpt-4o` by default, pinned with +`--judge-model`. Two rules the tool enforces or documents: - ```bash - poetry run python src/evaluation/evaluator.py \ - --testset_dir \ - --rag_type advanced \ - --model gpt-4o-mini - ``` +- **The judge may not be a model under test.** The tool refuses. A model grading + its own answers is not a measurement. +- **The judge must not change between runs being compared.** A moving judge makes + two runs incomparable, which is the failure this tool exists to avoid. - Responses and metric reports are written to `response//` and `evals//` inside the testset directory. +The judge's embedding model is only used to compare a question against an answer; +it is unrelated to the vectors in the bundle, and it is pointed at +`api.openai.com` explicitly rather than following `OPENAI_BASE_URL`, which on the +Plant Reactome host points at a self-hosted endpoint that does not serve it. +`JUDGE_BASE_URL` overrides. -Adjust the paths, model names, and distribution weights as needed for local experimentation. +## Reading the output +A single run has no noise floor: retrieval is not deterministic (Chroma's ANN +search varies run to run), so a difference between two single runs cannot be told +apart from variance. Use `--repeat 3` and compare against the reported spread. diff --git a/src/evaluation/evaluator.py b/src/evaluation/evaluator.py index b4ecf357..43f3e74d 100644 --- a/src/evaluation/evaluator.py +++ b/src/evaluation/evaluator.py @@ -1,229 +1,342 @@ +"""Score the answers the chatbot actually gives, with ragas. + +This measures **the shipping pipeline**. That is the whole point of the file and +was not true of it until now: it used to build its own `SelfQueryRetriever` + +`EnsembleRetriever` + `MergerRetriever` over the `summations` collection alone, +with `k=7` and weights `[0.2, 0.8]` that appear nowhere in the product. After the +retriever rewrite removed `SelfQueryRetriever` from the pipeline, it measured a +configuration that no longer existed anywhere -- producing numbers that looked +like an answer and were not. + +It now calls `create_reactome_rag`, the same factory `bin/chat-chainlit.py` +reaches, so a change to retrieval cannot alter the product without altering the +measurement. + + # one model over the golden questions + ./bin/evaluate --model gpt-4o-mini + + # two models, same questions, same judge, side by side + ./bin/evaluate --model gpt-4o-mini --model gpt-5.6-luna + + # how much of a difference is just noise? + ./bin/evaluate --model gpt-4o-mini --repeat 3 + +Requires an installed reactome bundle and OPENAI_API_KEY. +""" + import argparse +import json +import math import os +import statistics +import sys +import time from pathlib import Path from typing import Any -import pandas as pd -from datasets import Dataset -from langchain.retrievers import EnsembleRetriever -from langchain.retrievers.merger_retriever import MergerRetriever -from langchain.retrievers.self_query.base import SelfQueryRetriever -from langchain_chroma.vectorstores import Chroma -from langchain_community.document_loaders.csv_loader import CSVLoader -from langchain_community.retrievers import BM25Retriever -from langchain_core.retrievers import BaseRetriever -from langchain_core.runnables import Runnable -from langchain_openai import ChatOpenAI, OpenAIEmbeddings -from ragas import evaluate +import nltk +from dotenv import load_dotenv +from langchain_core.language_models.chat_models import BaseChatModel +from ragas import EvaluationDataset, SingleTurnSample, evaluate +from ragas.embeddings import LangchainEmbeddingsWrapper +from ragas.llms import LangchainLLMWrapper from ragas.metrics import ( ContextUtilization, - answer_relevancy, - context_recall, - faithfulness, + Faithfulness, + LLMContextRecall, + ResponseRelevancy, ) -from retrievers.rag_chain import create_rag_chain -from retrievers.reactome.metadata_info import ( - reactome_descriptions_info, - reactome_field_info, -) -from retrievers.reactome.prompt import reactome_qa_prompt +from agent.graph import resolve_embedding_model, resolve_temperature +from agent.models import get_embedding, get_llm +from retrievers.reactome.rag import create_reactome_rag from util.embedding_environment import EmbeddingEnvironment -context_utilization = ContextUtilization() +REPO_ROOT = Path(__file__).parent.parent.parent +DEFAULT_QUESTIONS = REPO_ROOT / "tests" / "golden" / "questions.txt" + +# The judge is pinned and is NOT the model under test. Scoring gpt-5.6-luna with +# gpt-5.6-luna would ask a model to grade its own homework; and a judge that moves +# between two runs makes the two runs incomparable, which is the failure this +# whole file exists to avoid. +DEFAULT_JUDGE_MODEL = "gpt-4o" + +# Used only to compare a question against an answer for answer_relevancy. +JUDGE_EMBEDDING_MODEL = "text-embedding-3-large" + + +def read_questions(path: Path) -> list[str]: + lines = path.read_text().splitlines() + return [ln.strip() for ln in lines if ln.strip() and not ln.startswith("#")] + + +def read_references(path: Path) -> dict[str, str]: + """Optional question -> reference answer map, as JSON. + + Only `context_recall` needs one. Without it that metric is dropped rather + than scored against nothing. + """ + data: dict[str, str] = json.loads(path.read_text()) + return data + + +def answer_questions( + chain: Any, questions: list[str] +) -> tuple[list[str], list[list[str]], float]: + """Ask the chain each question, keeping the answer and the retrieved context.""" + answers: list[str] = [] + contexts: list[list[str]] = [] + started = time.monotonic() + for i, question in enumerate(questions, start=1): + print(f" [{i}/{len(questions)}] {question[:70]}", file=sys.stderr) + response = chain.invoke({"input": question, "chat_history": []}) + answers.append(response["answer"]) + contexts.append([doc.page_content for doc in response["context"]]) + return answers, contexts, time.monotonic() - started + + +def build_chain(model: str, embeddings_dir: Path) -> Any: + """The chain under test, built exactly as the application builds it.""" + llm: BaseChatModel = get_llm( + "openai", + model, + request_timeout=360.0, + temperature=resolve_temperature(model), + ) + embedding = get_embedding("openai", resolve_embedding_model()) + return create_reactome_rag(llm, embedding, embeddings_dir) + + +def score( + questions: list[str], + answers: list[str], + contexts: list[list[str]], + references: dict[str, str], + judge: str, +) -> tuple[dict[str, float], list[dict[str, float]]]: + judge_llm = LangchainLLMWrapper( + get_llm("openai", judge, temperature=resolve_temperature(judge)) + ) + # answer_relevancy embeds the question and the answer to compare them. This + # is the JUDGE's embedding and has nothing to do with the vectors in the + # bundle, so it does not go through resolve_embedding_model -- but it must + # not silently follow OPENAI_BASE_URL either. On the Plant Reactome host that + # points at a self-hosted bge-m3 endpoint, and asking that for + # text-embedding-3-large is a 404 in the middle of a run. api.openai.com is + # named explicitly; JUDGE_BASE_URL overrides it. + judge_embeddings = LangchainEmbeddingsWrapper( + get_embedding( + "openai", + JUDGE_EMBEDDING_MODEL, + base_url=os.getenv("JUDGE_BASE_URL", "https://api.openai.com/v1"), + ) + ) + + samples = [ + SingleTurnSample( + user_input=q, + response=a, + retrieved_contexts=c, + reference=references.get(q), + ) + for q, a, c in zip(questions, answers, contexts, strict=True) + ] + + metrics: list[Any] = [Faithfulness(), ResponseRelevancy(), ContextUtilization()] + if all(s.reference for s in samples): + metrics.append(LLMContextRecall()) + else: + print( + " (no reference answers, so context_recall is skipped rather than " + "scored against nothing -- pass --references to include it)", + file=sys.stderr, + ) + + result = evaluate( + dataset=EvaluationDataset(samples=samples), + metrics=metrics, + llm=judge_llm, + embeddings=judge_embeddings, + ) + # result.scores is a public field holding one dict of metric -> score per + # sample. The aggregate used to come from result._repr_dict, which is private + # and would break on a ragas upgrade without warning -- in a file whose whole + # job is to stay trustworthy across upgrades. + per_question: list[dict[str, float]] = [dict(s) for s in result.scores] + + # A metric that fails on one sample comes back NaN, and NaN propagates + # through fmean -- so one bad question would turn the whole aggregate into + # NaN, which prints as "nan" and looks like a broken tool rather than a + # partial result. Non-finite scores are dropped and the drop is reported, + # because silently averaging over fewer questions than were asked is the kind + # of quiet difference this file is supposed to catch, not commit. + aggregate: dict[str, float] = {} + for metric in per_question[0]: + values = [ + s[metric] + for s in per_question + if s.get(metric) is not None and math.isfinite(s[metric]) + ] + dropped = len(per_question) - len(values) + if dropped: + print( + f" {metric}: {dropped} of {len(per_question)} questions could " + "not be scored and are excluded from the mean", + file=sys.stderr, + ) + aggregate[metric] = statistics.fmean(values) if values else float("nan") + return aggregate, per_question -def parse_arguments() -> argparse.Namespace: - """Parse command line arguments for the script.""" +def main() -> None: + load_dotenv() + try: + nltk.data.find("tokenizers/punkt_tab") + except LookupError: + nltk.download("punkt_tab", quiet=True) + parser = argparse.ArgumentParser( - description="Load a directory of testsets and evaluate answers generated by a language model." + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter ) parser.add_argument( - "--testset_dir", - type=str, - required=True, - help="Path to the directory containing testset Excel (.xlsx) files", + "--model", + action="append", + dest="models", + help="Model under test. Repeat to compare several against one judge.", ) parser.add_argument( - "--embeddings-dir", + "--judge-model", + default=DEFAULT_JUDGE_MODEL, + help=f"Model that scores the answers (default: {DEFAULT_JUDGE_MODEL}).", + ) + parser.add_argument("--questions", type=Path, default=DEFAULT_QUESTIONS) + parser.add_argument( + "--references", type=Path, - default=EmbeddingEnvironment.get_dir("reactome"), - help=( - "Reactome embeddings bundle to evaluate against. Defaults to the " - "installed one (see ./bin/embeddings_manager which)." - ), + help="JSON {question: reference answer}; enables context_recall.", ) parser.add_argument( - "--model", - type=str, - default="gpt-4o-mini", - help="Language model to use for evaluation", + "--repeat", + type=int, + default=1, + help="Run each model this many times, and report the spread as the noise " + "floor. A difference smaller than it is not a result.", ) + parser.add_argument("--limit", type=int, help="Use only the first N questions.") parser.add_argument( - "--rag_type", - choices=["basic", "advanced"], - required=True, - help="Type of RAG system to use for evaluation", + "--embeddings-dir", type=Path, default=EmbeddingEnvironment.get_dir("reactome") ) - return parser.parse_args() + parser.add_argument("--out", type=Path, help="Write the full report as JSON.") + args = parser.parse_args() + models: list[str] = args.models or ["gpt-4o-mini"] + if args.judge_model in models: + raise SystemExit( + f"The judge ({args.judge_model}) is also under test. A model grading " + "its own answers is not a measurement. Pass a different --judge-model." + ) + embeddings_dir: Path | None = args.embeddings_dir + if embeddings_dir is None: + raise SystemExit( + "No reactome embeddings installed. Run " + "./bin/embeddings_manager install , or pass " + "--embeddings-dir." + ) -def load_dataset(testset_path: str) -> list[dict[str, Any]]: - """Load the dataset from an Excel (.xlsx) file.""" - # pandas types record keys as Hashable; they are column names. - try: - df = pd.read_excel(testset_path) - records: list[dict[str, Any]] = df.to_dict(orient="records") # type: ignore[assignment] - return records - except FileNotFoundError as e: - raise FileNotFoundError(f"The file {testset_path} does not exist.") from e - except ValueError as e: - raise ValueError(f"Error reading the Excel file: {e}") from e - - -def initialize_rag_chain_with_memory( - embeddings_dir: Path, model_name: str, rag_type: str -) -> Runnable: - """Initialize the RAGChainWithMemory system. - - `embeddings_dir` is the bundle root, e.g. - embeddings/openai/text-embedding-3-large/reactome/Release95 . Both the BM25 - source CSV and the Chroma collection are derived from it; they used to be - absolute paths into a developer's home directory, so this script could not - run anywhere else. - """ - llm = ChatOpenAI(temperature=0.0, verbose=True, model=model_name) - retriever_list: list[BaseRetriever] = [] - - loader = CSVLoader(str(embeddings_dir / "csv_files" / "summations.csv")) - data = loader.load() - bm25_retriever = BM25Retriever.from_documents(data) - bm25_retriever.k = 7 - - # Set up vectorstore SelfQuery retriever - embedding = OpenAIEmbeddings(model="text-embedding-3-large") - vectordb = Chroma( - persist_directory=str(embeddings_dir / "summations"), - embedding_function=embedding, - ) - - vectordb_retriever = vectordb.as_retriever(search_kwargs={"k": 7}) + questions = read_questions(args.questions) + if args.limit: + questions = questions[: args.limit] + references = read_references(args.references) if args.references else {} - selfq_retriever = SelfQueryRetriever.from_llm( - llm=llm, - vectorstore=vectordb, - document_contents=reactome_descriptions_info["summations"], - metadata_field_info=reactome_field_info["summations"], - search_kwargs={"k": 7}, - ) - rrf_retriever = EnsembleRetriever( - retrievers=[bm25_retriever, selfq_retriever], weights=[0.2, 0.8] + print( + f"{len(questions)} questions x {len(models)} model(s) x {args.repeat} run(s), " + f"judged by {args.judge_model}\nbundle: {embeddings_dir}\n", + file=sys.stderr, ) - if rag_type == "basic": - retriever_list.append(vectordb_retriever) - elif rag_type == "advanced": - retriever_list.append(rrf_retriever) - reactome_retriever = MergerRetriever(retrievers=retriever_list) + report: dict[str, Any] = { + "judge_model": args.judge_model, + "bundle": str(embeddings_dir), + "questions": len(questions), + "repeat": args.repeat, + "models": {}, + } - return create_rag_chain( - retriever=reactome_retriever, - llm=llm, - qa_prompt=reactome_qa_prompt, - ) + for model in models: + runs: list[dict[str, float]] = [] + seconds: list[float] = [] + transcripts: list[list[dict[str, Any]]] = [] + chain = build_chain(model, embeddings_dir) + for run in range(1, args.repeat + 1): + print(f" {model} run {run}/{args.repeat}", file=sys.stderr) + answers, contexts, elapsed = answer_questions(chain, questions) + seconds.append(elapsed / len(questions)) + aggregate, per_question = score( + questions, answers, contexts, references, args.judge_model + ) + runs.append(aggregate) + # The answers themselves, not only the scores. The version this + # replaced wrote responses to a spreadsheet; dropping that would have + # left a low faithfulness score with nothing to look at to find out + # why. + transcripts.append( + [ + { + "question": q, + "answer": a, + "documents_retrieved": len(c), + "scores": s, + } + for q, a, c, s in zip( + questions, answers, contexts, per_question, strict=True + ) + ] + ) + report["models"][model] = { + "runs": runs, + "seconds_per_question": seconds, + "transcripts": transcripts, + } + print_report(report) + if args.out: + args.out.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n") + print(f"\nWrote {args.out}", file=sys.stderr) -def process_testset( - testset_path: str, - qa_system: Runnable, - response_dir: str, - eval_dir: str, - model_name: str, - rag_type: str, -) -> None: - """Process a single testset file.""" - testset = load_dataset(testset_path) - questions = [item["question"] for item in testset] - ground_truths = [item["ground_truth"] for item in testset] - - answers = [] - contexts = [] - - for question in questions: - response = qa_system.invoke({"input": question}) - answers.append(response["answer"]) - contexts.append([context.page_content for context in response["context"]]) - - rag_response_dir = os.path.join(response_dir, rag_type) - rag_eval_dir = os.path.join(eval_dir, rag_type) - os.makedirs(rag_response_dir, exist_ok=True) - os.makedirs(rag_eval_dir, exist_ok=True) - - # Save responses to an Excel file - data = { - "question": questions, - "answer": answers, - "contexts": contexts, - "ground_truth": ground_truths, - } - df_ans = pd.DataFrame(data) - response_filename = os.path.join( - rag_response_dir, - f"{os.path.splitext(os.path.basename(testset_path))[0]}_{model_name}_responses_{rag_type}.xlsx", - ) - df_ans.to_excel(response_filename, index=False) - print(f"Responses saved to {response_filename}") - # Evaluate the dataset - dataset = Dataset.from_dict(data) - result = evaluate( - llm=ChatOpenAI(temperature=0.0, verbose=True, model="gpt-4o"), - dataset=dataset, - metrics=[answer_relevancy, context_utilization, faithfulness, context_recall], - ) +def print_report(report: dict[str, Any]) -> None: + models: dict[str, Any] = report["models"] + metric_names = sorted({m for v in models.values() for m in v["runs"][0]}) - # Save evaluation results to an Excel file - evaluation_filename = os.path.join( - rag_eval_dir, - f"{os.path.splitext(os.path.basename(testset_path))[0]}_{model_name}_evaluation_{rag_type}.xlsx", + print( + f"\n{report['questions']} questions, judged by {report['judge_model']}, " + f"{report['repeat']} run(s) each\n" ) - df_eval = result.to_pandas() - df_eval.to_excel(evaluation_filename, index=False) - print(f"Evaluation results saved to {evaluation_filename}") - + header = f" {'model':<18}" + "".join(f"{m[:16]:>18}" for m in metric_names) + print(header + f"{'s/question':>13}") + print(" " + "-" * (len(header) + 11)) + for model, data in models.items(): + cells = "" + for metric in metric_names: + values = [r[metric] for r in data["runs"]] + mean = statistics.fmean(values) + spread = (max(values) - min(values)) if len(values) > 1 else 0.0 + cells += ( + f"{mean:>13.3f}±{spread:.3f}" if len(values) > 1 else f"{mean:>18.3f}" + ) + secs = statistics.fmean(data["seconds_per_question"]) + print(f" {model:<18}{cells}{secs:>13.1f}") -def main() -> None: - args = parse_arguments() - model_name = args.model - rag_type = args.rag_type - response_dir = os.path.join(args.testset_dir, "response") - eval_dir = os.path.join(args.testset_dir, "evals") - os.makedirs(response_dir, exist_ok=True) - os.makedirs(eval_dir, exist_ok=True) - - # Initialize RAG Chain - embeddings_dir: Path | None = args.embeddings_dir - if embeddings_dir is None: - raise SystemExit( - "No reactome embeddings installed and --embeddings-dir not given. " - "Install one with ./bin/embeddings_manager install ." + if report["repeat"] > 1: + print( + "\n The ± is the observed spread across runs -- the noise floor. A\n" + " difference between models smaller than it is not a result." + ) + else: + print( + "\n Single run, so there is no noise floor here and no way to tell a\n" + " real difference from run-to-run variance. Use --repeat 3 to compare." ) - qa_system = initialize_rag_chain_with_memory(embeddings_dir, model_name, rag_type) - - # Iterate over all .xlsx files in the directory - for filename in os.listdir(args.testset_dir): - print(f"Found file: {filename}") - if filename.endswith(".xlsx"): - testset_path = os.path.join(args.testset_dir, filename) - print(f"Processing testset: {testset_path}") - process_testset( - testset_path, - qa_system, - response_dir, - eval_dir, - model_name, - rag_type, - ) if __name__ == "__main__": diff --git a/tests/evaluation/test_evaluator.py b/tests/evaluation/test_evaluator.py new file mode 100644 index 00000000..aa0e2d27 --- /dev/null +++ b/tests/evaluation/test_evaluator.py @@ -0,0 +1,51 @@ +"""The parts of the evaluator that can be checked without spending an API call. + +The scoring itself needs the network and a bundle; what is pinned here is the +input handling and the guards -- including the one that stops a model from +grading its own answers, which would silently produce numbers rather than fail. +""" + +from pathlib import Path + +import pytest + +pytest.importorskip("ragas", reason="evaluation stack not installed") + +from evaluation.evaluator import ( # noqa: E402 + DEFAULT_JUDGE_MODEL, + DEFAULT_QUESTIONS, + read_questions, + read_references, +) + + +def test_the_default_question_set_is_the_committed_one() -> None: + """The evaluator and bin/retrieval_baseline must ask the same questions. + + Two measurement tools disagreeing about the question set would make their + results incomparable for no reason. + """ + assert DEFAULT_QUESTIONS.exists(), DEFAULT_QUESTIONS + assert DEFAULT_QUESTIONS.name == "questions.txt" + assert len(read_questions(DEFAULT_QUESTIONS)) >= 20 + + +def test_comments_and_blank_lines_are_not_questions(tmp_path: Path) -> None: + path = tmp_path / "q.txt" + path.write_text("# a heading\n\nWhat is TP53?\n\n#another\n Which complexes? \n") + assert read_questions(path) == ["What is TP53?", "Which complexes?"] + + +def test_references_are_read_as_a_question_to_answer_map(tmp_path: Path) -> None: + path = tmp_path / "refs.json" + path.write_text('{"What is TP53?": "A tumour suppressor."}') + assert read_references(path) == {"What is TP53?": "A tumour suppressor."} + + +def test_the_default_judge_is_not_a_model_this_repo_answers_with() -> None: + """The judge must be pinned and distinct; see the guard in main(). + + gpt-4o-mini is the default answering model, so the default judge must not be + it -- otherwise the out-of-the-box invocation grades its own homework. + """ + assert DEFAULT_JUDGE_MODEL != "gpt-4o-mini"