Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions bin/evaluate
Original file line number Diff line number Diff line change
@@ -1,6 +1,14 @@
#!/usr/bin/env python3
"""Entry point for the ragas evaluation; see src/evaluation/evaluator.py."""

import sys
from pathlib import Path

# Run from anywhere without setting PYTHONPATH. The Dockerfile and CI both set it
# (`/app/src`, `./bin:./src`), so importing worked there and nowhere else -- this
# script's own usage line, `./bin/evaluate`, raised ModuleNotFoundError.
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src"))

from evaluation.evaluator import main

if __name__ == "__main__":
Expand Down
34 changes: 29 additions & 5 deletions bin/probe_model_temperature
Original file line number Diff line number Diff line change
Expand Up @@ -11,15 +11,19 @@ This is the tool that produced it. Run it when a model is added, or when a 400
says a temperature is unsupported, and paste the printed set into graph.py. It
sends one one-token request per model.

./bin/probe_model_temperature # every chat model the key can see
./bin/probe_model_temperature gpt-6-nova # just these
poetry run ./bin/probe_model_temperature # every chat model the key can see
poetry run ./bin/probe_model_temperature gpt-6-nova # just these

Requires OPENAI_API_KEY.
"""

import os
import sys
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path

# Run from anywhere without setting PYTHONPATH; see bin/evaluate.
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "src"))

import openai
from dotenv import load_dotenv
Expand All @@ -28,9 +32,25 @@ from langchain_openai import ChatOpenAI
# Not chat models, or not reachable through /v1/chat/completions. Probing them
# yields 404s that say nothing about temperature.
NOT_CHAT = (
"audio", "image", "realtime", "transcribe", "tts", "whisper", "sora",
"moderation", "embedding", "search", "deep-research", "computer-use",
"instruct", "davinci", "babbage", "curie", "codex", "live", "-pro",
"audio",
"image",
"realtime",
"transcribe",
"tts",
"whisper",
"sora",
"moderation",
"embedding",
"search",
"deep-research",
"computer-use",
"instruct",
"davinci",
"babbage",
"curie",
"codex",
"live",
"-pro",
)


Expand All @@ -57,6 +77,10 @@ def main() -> None:
if not os.getenv("OPENAI_API_KEY"):
raise SystemExit("OPENAI_API_KEY is not set.")

if any(arg in ("-h", "--help") for arg in sys.argv[1:]):
print(__doc__)
return

client = openai.OpenAI()
models = sys.argv[1:] or sorted(
{m.id for m in client.models.list() if is_candidate(m.id)}
Expand Down
9 changes: 6 additions & 3 deletions src/evaluation/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -36,17 +36,20 @@ as JSON rather than from spreadsheets:
- An installed reactome bundle (`./bin/embeddings_manager install ...`)
- `OPENAI_API_KEY`

`poetry run` is needed for the interpreter, not for the import path: the script
puts `src/` on `sys.path` itself, so it works from any directory.

## Usage

```bash
# one model over the golden questions
./bin/evaluate --model gpt-4o-mini
poetry run ./bin/evaluate --model gpt-4o-mini

# two models, same questions, same judge, side by side
./bin/evaluate --model gpt-4o-mini --model gpt-5.6-luna
poetry run ./bin/evaluate --model gpt-4o-mini --model gpt-5.6-luna

# three runs each, so the report can show the noise floor
./bin/evaluate --model gpt-4o-mini --repeat 3 --out report.json
poetry run ./bin/evaluate --model gpt-4o-mini --repeat 3 --out report.json
```

`--out` writes the full report: aggregate scores per run, seconds per question,
Expand Down
Loading