Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 14 additions & 2 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -76,10 +76,22 @@ jobs:
shell: bash
run: |
base="${{ github.base_ref }}"
before="${{ github.event.before }}"
if [ -z "$base" ] && [ "${{ github.event_name }}" = "push" ]; then
# A push has no base branch; compare against what main was
# A push has no base branch; compare against where main was
# before it, so this stays as cheap here as it is on a PR.
if git diff --name-only HEAD^ HEAD -- poetry.lock | grep -q .; then
#
# github.event.before, not HEAD^. Rebase merges are enabled on
# this repository, so one push can advance main by several
# commits -- HEAD^ would then inspect only the last of them and
# miss a poetry.lock change in any earlier one, reporting
# changed=false and skipping the verification entirely.
# It is all zeros for the first push to a ref, and a force push
# can leave it unreachable, so fall back to verifying.
if [ -z "$before" ] || [ "$before" = "0000000000000000000000000000000000000000" ] \
|| ! git cat-file -e "$before^{commit}" 2>/dev/null; then
echo "changed=true" >> "$GITHUB_OUTPUT"
elif git diff --name-only "$before" HEAD -- poetry.lock | grep -q .; then
echo "changed=true" >> "$GITHUB_OUTPUT"
else
echo "changed=false" >> "$GITHUB_OUTPUT"
Expand Down
88 changes: 88 additions & 0 deletions bin/probe_model_temperature
Original file line number Diff line number Diff line change
@@ -0,0 +1,88 @@
#!/usr/bin/env python3
"""Re-derive which models refuse temperature=0.0.

`resolve_temperature` in src/agent/graph.py carries a list of models that accept
only their own default temperature of 1. No endpoint reports which values a model
takes, so that list is empirical -- and it interleaves in a way no name pattern
predicts: gpt-5 refuses 0.0, gpt-5.1/5.2/5.4 accept it, gpt-5.5 and gpt-5.6 refuse
it again.

This is the tool that produced it. Run it when a model is added, or when a 400
says a temperature is unsupported, and paste the printed set into graph.py. It
sends one one-token request per model.

./bin/probe_model_temperature # every chat model the key can see
./bin/probe_model_temperature gpt-6-nova # just these

Requires OPENAI_API_KEY.
"""

import os
import sys
from concurrent.futures import ThreadPoolExecutor

import openai
from dotenv import load_dotenv
from langchain_openai import ChatOpenAI

# Not chat models, or not reachable through /v1/chat/completions. Probing them
# yields 404s that say nothing about temperature.
NOT_CHAT = (
"audio", "image", "realtime", "transcribe", "tts", "whisper", "sora",
"moderation", "embedding", "search", "deep-research", "computer-use",
"instruct", "davinci", "babbage", "curie", "codex", "live", "-pro",
)


def is_candidate(model_id: str) -> bool:
return model_id.startswith(("gpt-", "o1", "o3", "o4", "chat-latest")) and not any(
skip in model_id for skip in NOT_CHAT
)


def probe(model: str) -> tuple[str, str]:
"""Send the smallest possible request and read the error, if any."""
try:
ChatOpenAI(model=model, temperature=0.0, max_tokens=1).invoke("hi")
return model, "accepts"
except Exception as exc: # noqa: BLE001 -- the error text is the result
text = str(exc)
if "does not support 0.0" in text:
return model, "refuses"
return model, "unknown"


def main() -> None:
load_dotenv()
if not os.getenv("OPENAI_API_KEY"):
raise SystemExit("OPENAI_API_KEY is not set.")

client = openai.OpenAI()
models = sys.argv[1:] or sorted(
{m.id for m in client.models.list() if is_candidate(m.id)}
)
print(f"Probing {len(models)} models with temperature=0.0...\n", file=sys.stderr)

with ThreadPoolExecutor(max_workers=8) as pool:
results = sorted(pool.map(probe, models))

for model, verdict in results:
print(f" {verdict:<9}{model}", file=sys.stderr)

refuses = sorted(m for m, v in results if v == "refuses")
unknown = sorted(m for m, v in results if v == "unknown")
if unknown:
print(
f"\n{len(unknown)} could not be probed (not a chat model, or no access): "
+ ", ".join(unknown),
file=sys.stderr,
)

print("\nPaste into FIXED_TEMPERATURE_MODELS in src/agent/graph.py,")
print("dropping any -YYYY-MM-DD snapshot suffix (those are handled there):\n")
for model in refuses:
print(f' "{model}",')


if __name__ == "__main__":
main()
38 changes: 32 additions & 6 deletions src/agent/graph.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import asyncio
import os
import re
from typing import Any, cast

from langchain_core.callbacks.base import Callbacks
Expand Down Expand Up @@ -89,13 +90,38 @@ def resolve_embedding_model() -> str:
# when the argument is omitted, and these models reject that too -- so the value
# has to be 1.0 explicitly.
#
# Unlike the embedding model, which is read from the bundle that built it, this
# cannot be derived: no endpoint reports which values a model accepts. Matched on
# prefix so dated snapshots (gpt-5.5-2026-04-23) and new members of a family
# need no edit. LLM_TEMPERATURE overrides, for a model this list has not met.
FIXED_TEMPERATURE_MODEL_PREFIXES = ("gpt-5.5", "gpt-5.6", "gpt-6")
# This is an exact-match set and not a name pattern, because the behaviour
# interleaves: gpt-5 refuses 0.0, gpt-5.1/5.2/5.4 accept it, and gpt-5.5/5.6
# refuse it again. A "gpt-5" prefix would have caught gpt-5.1 as well, and an
# earlier version of this file did exactly that -- and was wrong for eleven
# models, including the gpt-5 and o-series entries below.
#
# The list is empirical: no endpoint reports which values a model accepts, so it
# was measured. `./bin/probe_model_temperature` is the tool that measured it and
# prints this set; run it rather than reasoning about a name.
#
# Measured 2026-09-09. LLM_TEMPERATURE overrides, for a model added since.
FIXED_TEMPERATURE_MODELS = frozenset(
{
"chat-latest",
"gpt-5",
"gpt-5-mini",
"gpt-5-nano",
"gpt-5.5",
"gpt-5.6-luna",
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-6-astra",
"o3",
"o4-mini",
}
)
FIXED_TEMPERATURE = 1.0

# OpenAI pins dated snapshots of a model as `<model>-YYYY-MM-DD`. They behave as
# the model they pin, so the suffix is stripped rather than listing every one.
_SNAPSHOT_SUFFIX = re.compile(r"-\d{4}-\d{2}-\d{2}$")


def resolve_temperature(model: str) -> float:
"""The temperature to send for `model`.
Expand All @@ -109,7 +135,7 @@ def resolve_temperature(model: str) -> float:
return float(override)
except ValueError:
raise SystemExit(f"LLM_TEMPERATURE={override!r} is not a number.") from None
if model.startswith(FIXED_TEMPERATURE_MODEL_PREFIXES):
if _SNAPSHOT_SUFFIX.sub("", model) in FIXED_TEMPERATURE_MODELS:
return FIXED_TEMPERATURE
return 0.0

Expand Down
82 changes: 68 additions & 14 deletions tests/agent/test_model_temperature.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,11 @@

from langchain_openai.chat_models.base import ChatOpenAI # noqa: E402

from agent.graph import FIXED_TEMPERATURE, resolve_temperature # noqa: E402
from agent.graph import ( # noqa: E402
FIXED_TEMPERATURE,
FIXED_TEMPERATURE_MODELS,
resolve_temperature,
)
from agent.models import get_llm # noqa: E402


Expand All @@ -35,31 +39,81 @@ def _no_override(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.delenv("LLM_TEMPERATURE", raising=False)


@pytest.mark.parametrize(
"model",
["gpt-5.5", "gpt-5.5-2026-04-23", "gpt-5.6-luna", "gpt-5.6-sol", "gpt-6-astra"],
)
# Every verdict below was measured against the API by ./bin/probe_model_temperature
# on 2026-09-09, not inferred from the name. See test_the_set_is_not_a_name_pattern.
REFUSES_ZERO = [
"chat-latest",
"gpt-5",
"gpt-5-2025-08-07",
"gpt-5-mini",
"gpt-5-nano",
"gpt-5.5",
"gpt-5.5-2026-04-23",
"gpt-5.6-luna",
"gpt-5.6-sol",
"gpt-5.6-terra",
"gpt-6-astra",
"o3",
"o3-2025-04-16",
"o4-mini",
"o4-mini-2025-04-16",
]
ACCEPTS_ZERO = [
"gpt-3.5-turbo",
"gpt-4",
"gpt-4o-mini",
"gpt-4.1",
"gpt-4.1-nano",
"gpt-5.1",
"gpt-5.2",
"gpt-5.4",
"gpt-5.4-mini",
"gpt-5.4-nano-2026-03-17",
]


@pytest.mark.parametrize("model", REFUSES_ZERO)
def test_models_that_refuse_zero_get_their_only_supported_value(model: str) -> None:
assert resolve_temperature(model) == FIXED_TEMPERATURE == 1.0


@pytest.mark.parametrize("model", ["gpt-4o-mini", "gpt-4.1", "gpt-5", "gpt-5.4-mini"])
@pytest.mark.parametrize("model", ACCEPTS_ZERO)
def test_every_other_model_still_gets_zero(model: str) -> None:
"""Determinism stays the default; only the models that refuse it lose it.
"""Determinism stays the default; only the models that refuse it lose it."""
assert resolve_temperature(model) == 0.0


gpt-5 and gpt-5.4-mini are in this list deliberately: they are newer than
gpt-4o-mini and they do accept 0.0, so the rule is not "new models".
def test_the_set_is_not_a_name_pattern() -> None:
"""The reason this is an exact-match set and not a prefix match.

The behaviour interleaves inside one family: gpt-5 refuses 0.0, gpt-5.1,
gpt-5.2 and gpt-5.4 accept it, gpt-5.5 and gpt-5.6 refuse it again. A "gpt-5"
prefix -- which this file used to have -- also matches gpt-5.1, and was wrong
for eleven models.
"""
assert resolve_temperature(model) == 0.0
assert resolve_temperature("gpt-5") == FIXED_TEMPERATURE
for accepted in ("gpt-5.1", "gpt-5.2", "gpt-5.4"):
assert (
resolve_temperature(accepted) == 0.0
), f"{accepted} accepts 0.0; a gpt-5 prefix would have caught it"
assert resolve_temperature("gpt-5.5") == FIXED_TEMPERATURE


def test_the_prefix_table_covers_dated_snapshots() -> None:
"""gpt-5.6-luna and a dated pin of it must resolve the same way.
def test_dated_snapshots_resolve_as_the_model_they_pin() -> None:
"""`<model>-YYYY-MM-DD` behaves as `<model>`, so the suffix is stripped.

Matching on prefix is why this table needs no edit each time OpenAI pins a
snapshot of a family already listed.
This is what keeps the set from needing an edit every time OpenAI pins one.
"""
assert resolve_temperature("gpt-5.6-luna-2026-06-01") == FIXED_TEMPERATURE
assert resolve_temperature("gpt-5.4-mini-2026-03-17") == 0.0


def test_every_listed_model_is_bare_of_a_snapshot_suffix() -> None:
"""A dated id in the set would be dead: the suffix is stripped before lookup."""
for model in FIXED_TEMPERATURE_MODELS:
assert (
resolve_temperature(model) == FIXED_TEMPERATURE
), f"{model} is in the set but does not resolve through it"


def test_the_override_wins_over_the_table(monkeypatch: pytest.MonkeyPatch) -> None:
Expand Down
Loading