From cbe20fc6a1d8336a5e2de913a03f23d6f9817052 Mon Sep 17 00:00:00 2001 From: Adam Wright Date: Wed, 9 Sep 2026 16:48:25 +0000 Subject: [PATCH] Make bin/evaluate and bin/probe_model_temperature run as documented Both scripts landed today carrying a usage line -- `./bin/evaluate`, `./bin/probe_model_temperature` -- that raises ModuleNotFoundError. Imports resolved in the Dockerfile (PYTHONPATH=/app/src) and in CI (./bin:./src) and nowhere else, so the documented invocation was the one configuration nobody had run. Found by running the command in the README rather than the one used while developing. Each script now puts src/ on sys.path itself, from its own location, so it works from any directory. poetry run is still needed for the interpreter, and the README now says so. bin/probe_model_temperature also took --help as a model name and dutifully probed a model called "--help". It prints its docstring now. --- bin/evaluate | 8 ++++++++ bin/probe_model_temperature | 34 +++++++++++++++++++++++++++++----- src/evaluation/README.md | 9 ++++++--- 3 files changed, 43 insertions(+), 8 deletions(-) diff --git a/bin/evaluate b/bin/evaluate index 9bad3559..12078881 100755 --- a/bin/evaluate +++ b/bin/evaluate @@ -1,6 +1,14 @@ #!/usr/bin/env python3 """Entry point for the ragas evaluation; see src/evaluation/evaluator.py.""" +import sys +from pathlib import Path + +# Run from anywhere without setting PYTHONPATH. The Dockerfile and CI both set it +# (`/app/src`, `./bin:./src`), so importing worked there and nowhere else -- this +# script's own usage line, `./bin/evaluate`, raised ModuleNotFoundError. +sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src")) + from evaluation.evaluator import main if __name__ == "__main__": diff --git a/bin/probe_model_temperature b/bin/probe_model_temperature index 55b735ef..13bf35ef 100755 --- a/bin/probe_model_temperature +++ b/bin/probe_model_temperature @@ -11,8 +11,8 @@ This is the tool that produced it. Run it when a model is added, or when a 400 says a temperature is unsupported, and paste the printed set into graph.py. It sends one one-token request per model. - ./bin/probe_model_temperature # every chat model the key can see - ./bin/probe_model_temperature gpt-6-nova # just these + poetry run ./bin/probe_model_temperature # every chat model the key can see + poetry run ./bin/probe_model_temperature gpt-6-nova # just these Requires OPENAI_API_KEY. """ @@ -20,6 +20,10 @@ Requires OPENAI_API_KEY. import os import sys from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +# Run from anywhere without setting PYTHONPATH; see bin/evaluate. +sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src")) import openai from dotenv import load_dotenv @@ -28,9 +32,25 @@ from langchain_openai import ChatOpenAI # Not chat models, or not reachable through /v1/chat/completions. Probing them # yields 404s that say nothing about temperature. NOT_CHAT = ( - "audio", "image", "realtime", "transcribe", "tts", "whisper", "sora", - "moderation", "embedding", "search", "deep-research", "computer-use", - "instruct", "davinci", "babbage", "curie", "codex", "live", "-pro", + "audio", + "image", + "realtime", + "transcribe", + "tts", + "whisper", + "sora", + "moderation", + "embedding", + "search", + "deep-research", + "computer-use", + "instruct", + "davinci", + "babbage", + "curie", + "codex", + "live", + "-pro", ) @@ -57,6 +77,10 @@ def main() -> None: if not os.getenv("OPENAI_API_KEY"): raise SystemExit("OPENAI_API_KEY is not set.") + if any(arg in ("-h", "--help") for arg in sys.argv[1:]): + print(__doc__) + return + client = openai.OpenAI() models = sys.argv[1:] or sorted( {m.id for m in client.models.list() if is_candidate(m.id)} diff --git a/src/evaluation/README.md b/src/evaluation/README.md index 481b1c36..56e97ff7 100644 --- a/src/evaluation/README.md +++ b/src/evaluation/README.md @@ -36,17 +36,20 @@ as JSON rather than from spreadsheets: - An installed reactome bundle (`./bin/embeddings_manager install ...`) - `OPENAI_API_KEY` +`poetry run` is needed for the interpreter, not for the import path: the script +puts `src/` on `sys.path` itself, so it works from any directory. + ## Usage ```bash # one model over the golden questions -./bin/evaluate --model gpt-4o-mini +poetry run ./bin/evaluate --model gpt-4o-mini # two models, same questions, same judge, side by side -./bin/evaluate --model gpt-4o-mini --model gpt-5.6-luna +poetry run ./bin/evaluate --model gpt-4o-mini --model gpt-5.6-luna # three runs each, so the report can show the noise floor -./bin/evaluate --model gpt-4o-mini --repeat 3 --out report.json +poetry run ./bin/evaluate --model gpt-4o-mini --repeat 3 --out report.json ``` `--out` writes the full report: aggregate scores per run, seconds per question,