diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c332cd50..9ef33fd3 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -76,10 +76,22 @@ jobs: shell: bash run: | base="${{ github.base_ref }}" + before="${{ github.event.before }}" if [ -z "$base" ] && [ "${{ github.event_name }}" = "push" ]; then - # A push has no base branch; compare against what main was + # A push has no base branch; compare against where main was # before it, so this stays as cheap here as it is on a PR. - if git diff --name-only HEAD^ HEAD -- poetry.lock | grep -q .; then + # + # github.event.before, not HEAD^. Rebase merges are enabled on + # this repository, so one push can advance main by several + # commits -- HEAD^ would then inspect only the last of them and + # miss a poetry.lock change in any earlier one, reporting + # changed=false and skipping the verification entirely. + # It is all zeros for the first push to a ref, and a force push + # can leave it unreachable, so fall back to verifying. + if [ -z "$before" ] || [ "$before" = "0000000000000000000000000000000000000000" ] \ + || ! git cat-file -e "$before^{commit}" 2>/dev/null; then + echo "changed=true" >> "$GITHUB_OUTPUT" + elif git diff --name-only "$before" HEAD -- poetry.lock | grep -q .; then echo "changed=true" >> "$GITHUB_OUTPUT" else echo "changed=false" >> "$GITHUB_OUTPUT" diff --git a/bin/probe_model_temperature b/bin/probe_model_temperature new file mode 100755 index 00000000..55b735ef --- /dev/null +++ b/bin/probe_model_temperature @@ -0,0 +1,88 @@ +#!/usr/bin/env python3 +"""Re-derive which models refuse temperature=0.0. + +`resolve_temperature` in src/agent/graph.py carries a list of models that accept +only their own default temperature of 1. No endpoint reports which values a model +takes, so that list is empirical -- and it interleaves in a way no name pattern +predicts: gpt-5 refuses 0.0, gpt-5.1/5.2/5.4 accept it, gpt-5.5 and gpt-5.6 refuse +it again. + +This is the tool that produced it. Run it when a model is added, or when a 400 +says a temperature is unsupported, and paste the printed set into graph.py. It +sends one one-token request per model. + + ./bin/probe_model_temperature # every chat model the key can see + ./bin/probe_model_temperature gpt-6-nova # just these + +Requires OPENAI_API_KEY. +""" + +import os +import sys +from concurrent.futures import ThreadPoolExecutor + +import openai +from dotenv import load_dotenv +from langchain_openai import ChatOpenAI + +# Not chat models, or not reachable through /v1/chat/completions. Probing them +# yields 404s that say nothing about temperature. +NOT_CHAT = ( + "audio", "image", "realtime", "transcribe", "tts", "whisper", "sora", + "moderation", "embedding", "search", "deep-research", "computer-use", + "instruct", "davinci", "babbage", "curie", "codex", "live", "-pro", +) + + +def is_candidate(model_id: str) -> bool: + return model_id.startswith(("gpt-", "o1", "o3", "o4", "chat-latest")) and not any( + skip in model_id for skip in NOT_CHAT + ) + + +def probe(model: str) -> tuple[str, str]: + """Send the smallest possible request and read the error, if any.""" + try: + ChatOpenAI(model=model, temperature=0.0, max_tokens=1).invoke("hi") + return model, "accepts" + except Exception as exc: # noqa: BLE001 -- the error text is the result + text = str(exc) + if "does not support 0.0" in text: + return model, "refuses" + return model, "unknown" + + +def main() -> None: + load_dotenv() + if not os.getenv("OPENAI_API_KEY"): + raise SystemExit("OPENAI_API_KEY is not set.") + + client = openai.OpenAI() + models = sys.argv[1:] or sorted( + {m.id for m in client.models.list() if is_candidate(m.id)} + ) + print(f"Probing {len(models)} models with temperature=0.0...\n", file=sys.stderr) + + with ThreadPoolExecutor(max_workers=8) as pool: + results = sorted(pool.map(probe, models)) + + for model, verdict in results: + print(f" {verdict:<9}{model}", file=sys.stderr) + + refuses = sorted(m for m, v in results if v == "refuses") + unknown = sorted(m for m, v in results if v == "unknown") + if unknown: + print( + f"\n{len(unknown)} could not be probed (not a chat model, or no access): " + + ", ".join(unknown), + file=sys.stderr, + ) + + print("\nPaste into FIXED_TEMPERATURE_MODELS in src/agent/graph.py,") + print("dropping any -YYYY-MM-DD snapshot suffix (those are handled there):\n") + for model in refuses: + print(f' "{model}",') + + +if __name__ == "__main__": + main() diff --git a/src/agent/graph.py b/src/agent/graph.py index 4716ecdf..59e94da0 100644 --- a/src/agent/graph.py +++ b/src/agent/graph.py @@ -1,5 +1,6 @@ import asyncio import os +import re from typing import Any, cast from langchain_core.callbacks.base import Callbacks @@ -89,13 +90,38 @@ def resolve_embedding_model() -> str: # when the argument is omitted, and these models reject that too -- so the value # has to be 1.0 explicitly. # -# Unlike the embedding model, which is read from the bundle that built it, this -# cannot be derived: no endpoint reports which values a model accepts. Matched on -# prefix so dated snapshots (gpt-5.5-2026-04-23) and new members of a family -# need no edit. LLM_TEMPERATURE overrides, for a model this list has not met. -FIXED_TEMPERATURE_MODEL_PREFIXES = ("gpt-5.5", "gpt-5.6", "gpt-6") +# This is an exact-match set and not a name pattern, because the behaviour +# interleaves: gpt-5 refuses 0.0, gpt-5.1/5.2/5.4 accept it, and gpt-5.5/5.6 +# refuse it again. A "gpt-5" prefix would have caught gpt-5.1 as well, and an +# earlier version of this file did exactly that -- and was wrong for eleven +# models, including the gpt-5 and o-series entries below. +# +# The list is empirical: no endpoint reports which values a model accepts, so it +# was measured. `./bin/probe_model_temperature` is the tool that measured it and +# prints this set; run it rather than reasoning about a name. +# +# Measured 2026-09-09. LLM_TEMPERATURE overrides, for a model added since. +FIXED_TEMPERATURE_MODELS = frozenset( + { + "chat-latest", + "gpt-5", + "gpt-5-mini", + "gpt-5-nano", + "gpt-5.5", + "gpt-5.6-luna", + "gpt-5.6-sol", + "gpt-5.6-terra", + "gpt-6-astra", + "o3", + "o4-mini", + } +) FIXED_TEMPERATURE = 1.0 +# OpenAI pins dated snapshots of a model as `-YYYY-MM-DD`. They behave as +# the model they pin, so the suffix is stripped rather than listing every one. +_SNAPSHOT_SUFFIX = re.compile(r"-\d{4}-\d{2}-\d{2}$") + def resolve_temperature(model: str) -> float: """The temperature to send for `model`. @@ -109,7 +135,7 @@ def resolve_temperature(model: str) -> float: return float(override) except ValueError: raise SystemExit(f"LLM_TEMPERATURE={override!r} is not a number.") from None - if model.startswith(FIXED_TEMPERATURE_MODEL_PREFIXES): + if _SNAPSHOT_SUFFIX.sub("", model) in FIXED_TEMPERATURE_MODELS: return FIXED_TEMPERATURE return 0.0 diff --git a/tests/agent/test_model_temperature.py b/tests/agent/test_model_temperature.py index 33aa2493..1fd5c886 100644 --- a/tests/agent/test_model_temperature.py +++ b/tests/agent/test_model_temperature.py @@ -18,7 +18,11 @@ from langchain_openai.chat_models.base import ChatOpenAI # noqa: E402 -from agent.graph import FIXED_TEMPERATURE, resolve_temperature # noqa: E402 +from agent.graph import ( # noqa: E402 + FIXED_TEMPERATURE, + FIXED_TEMPERATURE_MODELS, + resolve_temperature, +) from agent.models import get_llm # noqa: E402 @@ -35,31 +39,81 @@ def _no_override(monkeypatch: pytest.MonkeyPatch) -> None: monkeypatch.delenv("LLM_TEMPERATURE", raising=False) -@pytest.mark.parametrize( - "model", - ["gpt-5.5", "gpt-5.5-2026-04-23", "gpt-5.6-luna", "gpt-5.6-sol", "gpt-6-astra"], -) +# Every verdict below was measured against the API by ./bin/probe_model_temperature +# on 2026-09-09, not inferred from the name. See test_the_set_is_not_a_name_pattern. +REFUSES_ZERO = [ + "chat-latest", + "gpt-5", + "gpt-5-2025-08-07", + "gpt-5-mini", + "gpt-5-nano", + "gpt-5.5", + "gpt-5.5-2026-04-23", + "gpt-5.6-luna", + "gpt-5.6-sol", + "gpt-5.6-terra", + "gpt-6-astra", + "o3", + "o3-2025-04-16", + "o4-mini", + "o4-mini-2025-04-16", +] +ACCEPTS_ZERO = [ + "gpt-3.5-turbo", + "gpt-4", + "gpt-4o-mini", + "gpt-4.1", + "gpt-4.1-nano", + "gpt-5.1", + "gpt-5.2", + "gpt-5.4", + "gpt-5.4-mini", + "gpt-5.4-nano-2026-03-17", +] + + +@pytest.mark.parametrize("model", REFUSES_ZERO) def test_models_that_refuse_zero_get_their_only_supported_value(model: str) -> None: assert resolve_temperature(model) == FIXED_TEMPERATURE == 1.0 -@pytest.mark.parametrize("model", ["gpt-4o-mini", "gpt-4.1", "gpt-5", "gpt-5.4-mini"]) +@pytest.mark.parametrize("model", ACCEPTS_ZERO) def test_every_other_model_still_gets_zero(model: str) -> None: - """Determinism stays the default; only the models that refuse it lose it. + """Determinism stays the default; only the models that refuse it lose it.""" + assert resolve_temperature(model) == 0.0 + - gpt-5 and gpt-5.4-mini are in this list deliberately: they are newer than - gpt-4o-mini and they do accept 0.0, so the rule is not "new models". +def test_the_set_is_not_a_name_pattern() -> None: + """The reason this is an exact-match set and not a prefix match. + + The behaviour interleaves inside one family: gpt-5 refuses 0.0, gpt-5.1, + gpt-5.2 and gpt-5.4 accept it, gpt-5.5 and gpt-5.6 refuse it again. A "gpt-5" + prefix -- which this file used to have -- also matches gpt-5.1, and was wrong + for eleven models. """ - assert resolve_temperature(model) == 0.0 + assert resolve_temperature("gpt-5") == FIXED_TEMPERATURE + for accepted in ("gpt-5.1", "gpt-5.2", "gpt-5.4"): + assert ( + resolve_temperature(accepted) == 0.0 + ), f"{accepted} accepts 0.0; a gpt-5 prefix would have caught it" + assert resolve_temperature("gpt-5.5") == FIXED_TEMPERATURE -def test_the_prefix_table_covers_dated_snapshots() -> None: - """gpt-5.6-luna and a dated pin of it must resolve the same way. +def test_dated_snapshots_resolve_as_the_model_they_pin() -> None: + """`-YYYY-MM-DD` behaves as ``, so the suffix is stripped. - Matching on prefix is why this table needs no edit each time OpenAI pins a - snapshot of a family already listed. + This is what keeps the set from needing an edit every time OpenAI pins one. """ assert resolve_temperature("gpt-5.6-luna-2026-06-01") == FIXED_TEMPERATURE + assert resolve_temperature("gpt-5.4-mini-2026-03-17") == 0.0 + + +def test_every_listed_model_is_bare_of_a_snapshot_suffix() -> None: + """A dated id in the set would be dead: the suffix is stripped before lookup.""" + for model in FIXED_TEMPERATURE_MODELS: + assert ( + resolve_temperature(model) == FIXED_TEMPERATURE + ), f"{model} is in the set but does not resolve through it" def test_the_override_wins_over_the_table(monkeypatch: pytest.MonkeyPatch) -> None: