From c0d3509d0713129701b86f2ecfdb497365363f3e Mon Sep 17 00:00:00 2001 From: Adam Wright Date: Wed, 9 Sep 2026 14:22:52 +0000 Subject: [PATCH] Make gpt-5.6-luna usable: per-model temperature, json_schema graders Two things in this repository stopped gpt-5.6-luna working at all. Both are fixed here; the default model is unchanged. 1. Temperature. get_llm hardcoded temperature=0.0, which this repo wants everywhere -- the graders, the intent classifier and the query expander should give the same answer twice. The gpt-5.5/5.6 families and gpt-6-astra reject it: Unsupported value: 'temperature' does not support 0.0 with this model. Only the default (1) value is supported. as a 400 on the first request, not at construction, so nothing notices until a user asks a question. resolve_temperature() sends 1.0 for those families and 0.0 for everything else, with LLM_TEMPERATURE to override. Sending *no* temperature is not an option, and this was the trap: on langchain-openai 0.2.14, omitting the argument makes ChatOpenAI send its own pydantic default of 0.7, which these models refuse just as firmly. The first version of this commit did exactly that, and model_dump(exclude_unset=True) confirmed the field was unset while the request still carried 0.7. Only running it against the API showed it. Hence a value, not an omission. Temperature 1 costs determinism. Measured over 10 runs each on four inputs -- science question, how-to question, benign text, prompt injection -- both the intent classifier and the safety checker returned the same verdict every time on gpt-5.6-luna as on gpt-4o-mini at 0.0. Stable on what was tested; not a guarantee. 2. Structured output. The three graders used the default function_calling method, which the gpt-5.6 family refuses on /v1/chat/completions ("Function tools with reasoning_effort are not supported ... use /v1/responses or set reasoning_effort to 'none'"), and langchain-openai 0.2.14 has no Responses API support. method="json_schema" uses response_format instead. Verified for all three graders against both gpt-4o-mini and gpt-5.6-luna, so this is not a luna-only path that would rot untested. Verified end to end through AgentGraph.ainvoke on the React-to-Me profile, the same entry point bin/chat-chainlit.py uses, with the Release95 bundle and three real questions. Both models answer. Averaged per question: gpt-4o-mini 22.5s, gpt-5.6-luna 41.2s, 6 LLM calls each. --- deploy/beta/env.beta.template | 8 ++ env_template | 8 ++ src/agent/graph.py | 42 +++++++++- src/agent/models.py | 17 ++++- src/agent/tasks/completeness_grader.py | 10 ++- src/agent/tasks/intent_classifier.py | 10 ++- src/agent/tasks/safety_checker.py | 10 ++- tests/agent/test_model_temperature.py | 102 +++++++++++++++++++++++++ 8 files changed, 201 insertions(+), 6 deletions(-) create mode 100644 tests/agent/test_model_temperature.py diff --git a/deploy/beta/env.beta.template b/deploy/beta/env.beta.template index 8a2f7e26..13e11b5f 100644 --- a/deploy/beta/env.beta.template +++ b/deploy/beta/env.beta.template @@ -22,4 +22,12 @@ CHAINLIT_URL=https://beta.reactome.org # anything else makes retrieval meaningless. #EMBEDDING_MODEL= # LLM used for answers; defaults to gpt-4o-mini. +# gpt-5.6-luna is verified to work end to end. It answers with noticeably more +# grounding in the retrieved Reactome records, and takes roughly twice as long +# per question (41s vs 22s, averaged over three questions). #LLM_MODEL= +# Sampling temperature. Leave unset: it defaults to 0.0, except for the +# gpt-5.5/5.6/6 families, which accept only their own default of 1 and return a +# 400 for anything else. Set this only for a model that rule does not yet know +# about; see resolve_temperature in src/agent/graph.py. +#LLM_TEMPERATURE= diff --git a/env_template b/env_template index e0963807..a7343c9e 100644 --- a/env_template +++ b/env_template @@ -20,4 +20,12 @@ TAVILY_API_KEY= # anything else makes retrieval meaningless. #EMBEDDING_MODEL= # LLM used for answers; defaults to gpt-4o-mini. +# gpt-5.6-luna is verified to work end to end. It answers with noticeably more +# grounding in the retrieved Reactome records, and takes roughly twice as long +# per question (41s vs 22s, averaged over three questions). #LLM_MODEL= +# Sampling temperature. Leave unset: it defaults to 0.0, except for the +# gpt-5.5/5.6/6 families, which accept only their own default of 1 and return a +# 400 for anything else. Set this only for a model that rule does not yet know +# about; see resolve_temperature in src/agent/graph.py. +#LLM_TEMPERATURE= diff --git a/src/agent/graph.py b/src/agent/graph.py index 4fafd9db..4716ecdf 100644 --- a/src/agent/graph.py +++ b/src/agent/graph.py @@ -78,6 +78,42 @@ def resolve_embedding_model() -> str: return configured or bundle_model +# Models that accept only one temperature: their own default of 1. Any other +# value, 0.0 included, is a 400 on the first request rather than an error at +# construction: +# +# Unsupported value: 'temperature' does not support 0.0 with this model. +# Only the default (1) value is supported. +# +# Sending nothing is not an option -- ChatOpenAI supplies its own default of 0.7 +# when the argument is omitted, and these models reject that too -- so the value +# has to be 1.0 explicitly. +# +# Unlike the embedding model, which is read from the bundle that built it, this +# cannot be derived: no endpoint reports which values a model accepts. Matched on +# prefix so dated snapshots (gpt-5.5-2026-04-23) and new members of a family +# need no edit. LLM_TEMPERATURE overrides, for a model this list has not met. +FIXED_TEMPERATURE_MODEL_PREFIXES = ("gpt-5.5", "gpt-5.6", "gpt-6") +FIXED_TEMPERATURE = 1.0 + + +def resolve_temperature(model: str) -> float: + """The temperature to send for `model`. + + Returning 1.0 for the models above trades determinism for being able to use + them at all. That trade is made here, once, rather than at each call site. + """ + override = os.getenv("LLM_TEMPERATURE") + if override is not None and override.strip() != "": + try: + return float(override) + except ValueError: + raise SystemExit(f"LLM_TEMPERATURE={override!r} is not a number.") from None + if model.startswith(FIXED_TEMPERATURE_MODEL_PREFIXES): + return FIXED_TEMPERATURE + return 0.0 + + class AgentGraph: def __init__( self, @@ -88,7 +124,11 @@ def __init__( llm_model = os.getenv("LLM_MODEL", "gpt-4o-mini") llm_base_url = os.getenv("LLM_BASE_URL", None) llm: BaseChatModel = get_llm( - "openai", llm_model, base_url=llm_base_url, request_timeout=360.0 + "openai", + llm_model, + base_url=llm_base_url, + request_timeout=360.0, + temperature=resolve_temperature(llm_model), ) embedding_base_url = os.getenv("OPENAI_BASE_URL", None) embedding: Embeddings = get_embedding( diff --git a/src/agent/models.py b/src/agent/models.py index 904ca3e4..cf7f74c9 100644 --- a/src/agent/models.py +++ b/src/agent/models.py @@ -49,20 +49,33 @@ def get_llm( *, base_url: str | None = None, request_timeout: float | None = None, + temperature: float = 0.0, ) -> BaseChatModel: + """Build a chat model. See `agent.graph.resolve_temperature` for the value. + + 0.0 is what this repository wants everywhere -- the graders, the intent + classifier and the query expander should all give the same answer twice. + Some models refuse it, which is why this is a parameter rather than the + constant it used to be. + + There is no way to send *no* temperature on langchain-openai 0.2.14: + omitting the argument makes ChatOpenAI send its own pydantic default of 0.7, + which the models that refuse 0.0 refuse just as firmly. So the caller must + pass a value the model accepts; it cannot opt out. + """ if model is None: provider, model = provider.split("/", 1) if provider == "openai": return ChatOpenAI( model=model, - temperature=0.0, + temperature=temperature, base_url=base_url, request_timeout=request_timeout, # type: ignore[call-arg] # pydantic-generated __init__ ) if provider == "ollama": return ChatOllama( model=model, - temperature=0.0, + temperature=temperature, base_url=base_url, request_timeout=request_timeout, # type: ignore[call-arg] # pydantic-generated __init__ ) diff --git a/src/agent/tasks/completeness_grader.py b/src/agent/tasks/completeness_grader.py index e2254ac8..264c3717 100644 --- a/src/agent/tasks/completeness_grader.py +++ b/src/agent/tasks/completeness_grader.py @@ -28,5 +28,13 @@ class CompletenessGrade(BaseModel): ) +# json_schema, not the default function_calling: the gpt-5.6 family refuses +# function tools on /v1/chat/completions ("Function tools with reasoning_effort +# are not supported ... use /v1/responses or set reasoning_effort to 'none'"), +# and langchain-openai 0.2.14 has no Responses API support. json_schema uses +# response_format instead, which every model here accepts -- verified against +# gpt-4o-mini and gpt-5.6-luna for all three graders. def create_completeness_grader(llm: BaseChatModel) -> Runnable: - return completeness_prompt | llm.with_structured_output(CompletenessGrade) + return completeness_prompt | llm.with_structured_output( + CompletenessGrade, method="json_schema" + ) diff --git a/src/agent/tasks/intent_classifier.py b/src/agent/tasks/intent_classifier.py index 17aa2b0f..bf1d72bc 100644 --- a/src/agent/tasks/intent_classifier.py +++ b/src/agent/tasks/intent_classifier.py @@ -57,5 +57,13 @@ def resolve_active_sources( return [next(iter(available_sources))] +# json_schema, not the default function_calling: the gpt-5.6 family refuses +# function tools on /v1/chat/completions ("Function tools with reasoning_effort +# are not supported ... use /v1/responses or set reasoning_effort to 'none'"), +# and langchain-openai 0.2.14 has no Responses API support. json_schema uses +# response_format instead, which every model here accepts -- verified against +# gpt-4o-mini and gpt-5.6-luna for all three graders. def create_intent_classifier(llm: BaseChatModel) -> Runnable: - return intent_classifier_prompt | llm.with_structured_output(QueryIntent) + return intent_classifier_prompt | llm.with_structured_output( + QueryIntent, method="json_schema" + ) diff --git a/src/agent/tasks/safety_checker.py b/src/agent/tasks/safety_checker.py index c1360133..f69fcadb 100644 --- a/src/agent/tasks/safety_checker.py +++ b/src/agent/tasks/safety_checker.py @@ -69,5 +69,13 @@ class SafetyCheck(BaseModel): ) +# json_schema, not the default function_calling: the gpt-5.6 family refuses +# function tools on /v1/chat/completions ("Function tools with reasoning_effort +# are not supported ... use /v1/responses or set reasoning_effort to 'none'"), +# and langchain-openai 0.2.14 has no Responses API support. json_schema uses +# response_format instead, which every model here accepts -- verified against +# gpt-4o-mini and gpt-5.6-luna for all three graders. def create_safety_checker(llm: BaseChatModel) -> Runnable: - return safety_check_prompt | llm.with_structured_output(SafetyCheck) + return safety_check_prompt | llm.with_structured_output( + SafetyCheck, method="json_schema" + ) diff --git a/tests/agent/test_model_temperature.py b/tests/agent/test_model_temperature.py new file mode 100644 index 00000000..33aa2493 --- /dev/null +++ b/tests/agent/test_model_temperature.py @@ -0,0 +1,102 @@ +"""Which temperature gets sent, and to which models. + +The gpt-5.5/5.6 families and gpt-6-astra accept only their own default of 1. +Every other value is a 400 on the *first request*, not at construction, so +nothing catches it until a user asks a question. + +There is no way to send no temperature at all: omitting the argument makes +ChatOpenAI send its own pydantic default of 0.7, which those models reject just +as firmly as 0.0. An earlier version of this change tried exactly that, and +`model_dump(exclude_unset=True)` agreed the field was unset -- while the request +still carried 0.7. So the assertions below are about the value chosen, and the +live behaviour was verified by hand against the API. +""" + +import pytest + +pytest.importorskip("langchain_openai", reason="LLM stack not installed") + +from langchain_openai.chat_models.base import ChatOpenAI # noqa: E402 + +from agent.graph import FIXED_TEMPERATURE, resolve_temperature # noqa: E402 +from agent.models import get_llm # noqa: E402 + + +def _built(model: str, **kwargs: float) -> ChatOpenAI: + """get_llm is typed BaseChatModel; the temperature lives on ChatOpenAI.""" + llm = get_llm("openai", model, **kwargs) # type: ignore[arg-type] + assert isinstance(llm, ChatOpenAI) + return llm + + +@pytest.fixture(autouse=True) +def _no_override(monkeypatch: pytest.MonkeyPatch) -> None: + """A developer's LLM_TEMPERATURE must not decide what these tests assert.""" + monkeypatch.delenv("LLM_TEMPERATURE", raising=False) + + +@pytest.mark.parametrize( + "model", + ["gpt-5.5", "gpt-5.5-2026-04-23", "gpt-5.6-luna", "gpt-5.6-sol", "gpt-6-astra"], +) +def test_models_that_refuse_zero_get_their_only_supported_value(model: str) -> None: + assert resolve_temperature(model) == FIXED_TEMPERATURE == 1.0 + + +@pytest.mark.parametrize("model", ["gpt-4o-mini", "gpt-4.1", "gpt-5", "gpt-5.4-mini"]) +def test_every_other_model_still_gets_zero(model: str) -> None: + """Determinism stays the default; only the models that refuse it lose it. + + gpt-5 and gpt-5.4-mini are in this list deliberately: they are newer than + gpt-4o-mini and they do accept 0.0, so the rule is not "new models". + """ + assert resolve_temperature(model) == 0.0 + + +def test_the_prefix_table_covers_dated_snapshots() -> None: + """gpt-5.6-luna and a dated pin of it must resolve the same way. + + Matching on prefix is why this table needs no edit each time OpenAI pins a + snapshot of a family already listed. + """ + assert resolve_temperature("gpt-5.6-luna-2026-06-01") == FIXED_TEMPERATURE + + +def test_the_override_wins_over_the_table(monkeypatch: pytest.MonkeyPatch) -> None: + """The escape hatch, for a model the table has not met.""" + monkeypatch.setenv("LLM_TEMPERATURE", "1") + assert resolve_temperature("gpt-4o-mini") == 1.0 + monkeypatch.setenv("LLM_TEMPERATURE", "0") + assert resolve_temperature("gpt-5.6-luna") == 0.0 + + +@pytest.mark.parametrize("value", ["", " "]) +def test_an_empty_override_is_treated_as_unset( + value: str, monkeypatch: pytest.MonkeyPatch +) -> None: + """An empty env var in a compose file means "not configured", not 0.0.""" + monkeypatch.setenv("LLM_TEMPERATURE", value) + assert resolve_temperature("gpt-4o-mini") == 0.0 + + +def test_an_unparseable_override_stops_the_process( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Article IV: a temperature nobody can honour must not fall back silently.""" + monkeypatch.setenv("LLM_TEMPERATURE", "warm") + with pytest.raises(SystemExit, match="not a number"): + resolve_temperature("gpt-4o-mini") + + +def test_get_llm_sends_the_temperature_it_is_given( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setenv("OPENAI_API_KEY", "sk-not-a-real-key") # construction only + assert _built("gpt-4o-mini", temperature=0.0).temperature == 0.0 + assert _built("gpt-5.6-luna", temperature=1.0).temperature == 1.0 + + +def test_get_llm_still_defaults_to_zero(monkeypatch: pytest.MonkeyPatch) -> None: + """Callers that never heard of this change keep the old behaviour.""" + monkeypatch.setenv("OPENAI_API_KEY", "sk-not-a-real-key") + assert _built("gpt-4o-mini").temperature == 0.0