diff --git a/deploy/beta/env.beta.template b/deploy/beta/env.beta.template index 8a2f7e26..13e11b5f 100644 --- a/deploy/beta/env.beta.template +++ b/deploy/beta/env.beta.template @@ -22,4 +22,12 @@ CHAINLIT_URL=https://beta.reactome.org # anything else makes retrieval meaningless. #EMBEDDING_MODEL= # LLM used for answers; defaults to gpt-4o-mini. +# gpt-5.6-luna is verified to work end to end. It answers with noticeably more +# grounding in the retrieved Reactome records, and takes roughly twice as long +# per question (41s vs 22s, averaged over three questions). #LLM_MODEL= +# Sampling temperature. Leave unset: it defaults to 0.0, except for the +# gpt-5.5/5.6/6 families, which accept only their own default of 1 and return a +# 400 for anything else. Set this only for a model that rule does not yet know +# about; see resolve_temperature in src/agent/graph.py. +#LLM_TEMPERATURE= diff --git a/env_template b/env_template index e0963807..a7343c9e 100644 --- a/env_template +++ b/env_template @@ -20,4 +20,12 @@ TAVILY_API_KEY= # anything else makes retrieval meaningless. #EMBEDDING_MODEL= # LLM used for answers; defaults to gpt-4o-mini. +# gpt-5.6-luna is verified to work end to end. It answers with noticeably more +# grounding in the retrieved Reactome records, and takes roughly twice as long +# per question (41s vs 22s, averaged over three questions). #LLM_MODEL= +# Sampling temperature. Leave unset: it defaults to 0.0, except for the +# gpt-5.5/5.6/6 families, which accept only their own default of 1 and return a +# 400 for anything else. Set this only for a model that rule does not yet know +# about; see resolve_temperature in src/agent/graph.py. +#LLM_TEMPERATURE= diff --git a/src/agent/graph.py b/src/agent/graph.py index 4fafd9db..4716ecdf 100644 --- a/src/agent/graph.py +++ b/src/agent/graph.py @@ -78,6 +78,42 @@ def resolve_embedding_model() -> str: return configured or bundle_model +# Models that accept only one temperature: their own default of 1. Any other +# value, 0.0 included, is a 400 on the first request rather than an error at +# construction: +# +# Unsupported value: 'temperature' does not support 0.0 with this model. +# Only the default (1) value is supported. +# +# Sending nothing is not an option -- ChatOpenAI supplies its own default of 0.7 +# when the argument is omitted, and these models reject that too -- so the value +# has to be 1.0 explicitly. +# +# Unlike the embedding model, which is read from the bundle that built it, this +# cannot be derived: no endpoint reports which values a model accepts. Matched on +# prefix so dated snapshots (gpt-5.5-2026-04-23) and new members of a family +# need no edit. LLM_TEMPERATURE overrides, for a model this list has not met. +FIXED_TEMPERATURE_MODEL_PREFIXES = ("gpt-5.5", "gpt-5.6", "gpt-6") +FIXED_TEMPERATURE = 1.0 + + +def resolve_temperature(model: str) -> float: + """The temperature to send for `model`. + + Returning 1.0 for the models above trades determinism for being able to use + them at all. That trade is made here, once, rather than at each call site. + """ + override = os.getenv("LLM_TEMPERATURE") + if override is not None and override.strip() != "": + try: + return float(override) + except ValueError: + raise SystemExit(f"LLM_TEMPERATURE={override!r} is not a number.") from None + if model.startswith(FIXED_TEMPERATURE_MODEL_PREFIXES): + return FIXED_TEMPERATURE + return 0.0 + + class AgentGraph: def __init__( self, @@ -88,7 +124,11 @@ def __init__( llm_model = os.getenv("LLM_MODEL", "gpt-4o-mini") llm_base_url = os.getenv("LLM_BASE_URL", None) llm: BaseChatModel = get_llm( - "openai", llm_model, base_url=llm_base_url, request_timeout=360.0 + "openai", + llm_model, + base_url=llm_base_url, + request_timeout=360.0, + temperature=resolve_temperature(llm_model), ) embedding_base_url = os.getenv("OPENAI_BASE_URL", None) embedding: Embeddings = get_embedding( diff --git a/src/agent/models.py b/src/agent/models.py index 904ca3e4..cf7f74c9 100644 --- a/src/agent/models.py +++ b/src/agent/models.py @@ -49,20 +49,33 @@ def get_llm( *, base_url: str | None = None, request_timeout: float | None = None, + temperature: float = 0.0, ) -> BaseChatModel: + """Build a chat model. See `agent.graph.resolve_temperature` for the value. + + 0.0 is what this repository wants everywhere -- the graders, the intent + classifier and the query expander should all give the same answer twice. + Some models refuse it, which is why this is a parameter rather than the + constant it used to be. + + There is no way to send *no* temperature on langchain-openai 0.2.14: + omitting the argument makes ChatOpenAI send its own pydantic default of 0.7, + which the models that refuse 0.0 refuse just as firmly. So the caller must + pass a value the model accepts; it cannot opt out. + """ if model is None: provider, model = provider.split("/", 1) if provider == "openai": return ChatOpenAI( model=model, - temperature=0.0, + temperature=temperature, base_url=base_url, request_timeout=request_timeout, # type: ignore[call-arg] # pydantic-generated __init__ ) if provider == "ollama": return ChatOllama( model=model, - temperature=0.0, + temperature=temperature, base_url=base_url, request_timeout=request_timeout, # type: ignore[call-arg] # pydantic-generated __init__ ) diff --git a/src/agent/tasks/completeness_grader.py b/src/agent/tasks/completeness_grader.py index e2254ac8..264c3717 100644 --- a/src/agent/tasks/completeness_grader.py +++ b/src/agent/tasks/completeness_grader.py @@ -28,5 +28,13 @@ class CompletenessGrade(BaseModel): ) +# json_schema, not the default function_calling: the gpt-5.6 family refuses +# function tools on /v1/chat/completions ("Function tools with reasoning_effort +# are not supported ... use /v1/responses or set reasoning_effort to 'none'"), +# and langchain-openai 0.2.14 has no Responses API support. json_schema uses +# response_format instead, which every model here accepts -- verified against +# gpt-4o-mini and gpt-5.6-luna for all three graders. def create_completeness_grader(llm: BaseChatModel) -> Runnable: - return completeness_prompt | llm.with_structured_output(CompletenessGrade) + return completeness_prompt | llm.with_structured_output( + CompletenessGrade, method="json_schema" + ) diff --git a/src/agent/tasks/intent_classifier.py b/src/agent/tasks/intent_classifier.py index 17aa2b0f..bf1d72bc 100644 --- a/src/agent/tasks/intent_classifier.py +++ b/src/agent/tasks/intent_classifier.py @@ -57,5 +57,13 @@ def resolve_active_sources( return [next(iter(available_sources))] +# json_schema, not the default function_calling: the gpt-5.6 family refuses +# function tools on /v1/chat/completions ("Function tools with reasoning_effort +# are not supported ... use /v1/responses or set reasoning_effort to 'none'"), +# and langchain-openai 0.2.14 has no Responses API support. json_schema uses +# response_format instead, which every model here accepts -- verified against +# gpt-4o-mini and gpt-5.6-luna for all three graders. def create_intent_classifier(llm: BaseChatModel) -> Runnable: - return intent_classifier_prompt | llm.with_structured_output(QueryIntent) + return intent_classifier_prompt | llm.with_structured_output( + QueryIntent, method="json_schema" + ) diff --git a/src/agent/tasks/safety_checker.py b/src/agent/tasks/safety_checker.py index c1360133..f69fcadb 100644 --- a/src/agent/tasks/safety_checker.py +++ b/src/agent/tasks/safety_checker.py @@ -69,5 +69,13 @@ class SafetyCheck(BaseModel): ) +# json_schema, not the default function_calling: the gpt-5.6 family refuses +# function tools on /v1/chat/completions ("Function tools with reasoning_effort +# are not supported ... use /v1/responses or set reasoning_effort to 'none'"), +# and langchain-openai 0.2.14 has no Responses API support. json_schema uses +# response_format instead, which every model here accepts -- verified against +# gpt-4o-mini and gpt-5.6-luna for all three graders. def create_safety_checker(llm: BaseChatModel) -> Runnable: - return safety_check_prompt | llm.with_structured_output(SafetyCheck) + return safety_check_prompt | llm.with_structured_output( + SafetyCheck, method="json_schema" + ) diff --git a/tests/agent/test_model_temperature.py b/tests/agent/test_model_temperature.py new file mode 100644 index 00000000..33aa2493 --- /dev/null +++ b/tests/agent/test_model_temperature.py @@ -0,0 +1,102 @@ +"""Which temperature gets sent, and to which models. + +The gpt-5.5/5.6 families and gpt-6-astra accept only their own default of 1. +Every other value is a 400 on the *first request*, not at construction, so +nothing catches it until a user asks a question. + +There is no way to send no temperature at all: omitting the argument makes +ChatOpenAI send its own pydantic default of 0.7, which those models reject just +as firmly as 0.0. An earlier version of this change tried exactly that, and +`model_dump(exclude_unset=True)` agreed the field was unset -- while the request +still carried 0.7. So the assertions below are about the value chosen, and the +live behaviour was verified by hand against the API. +""" + +import pytest + +pytest.importorskip("langchain_openai", reason="LLM stack not installed") + +from langchain_openai.chat_models.base import ChatOpenAI # noqa: E402 + +from agent.graph import FIXED_TEMPERATURE, resolve_temperature # noqa: E402 +from agent.models import get_llm # noqa: E402 + + +def _built(model: str, **kwargs: float) -> ChatOpenAI: + """get_llm is typed BaseChatModel; the temperature lives on ChatOpenAI.""" + llm = get_llm("openai", model, **kwargs) # type: ignore[arg-type] + assert isinstance(llm, ChatOpenAI) + return llm + + +@pytest.fixture(autouse=True) +def _no_override(monkeypatch: pytest.MonkeyPatch) -> None: + """A developer's LLM_TEMPERATURE must not decide what these tests assert.""" + monkeypatch.delenv("LLM_TEMPERATURE", raising=False) + + +@pytest.mark.parametrize( + "model", + ["gpt-5.5", "gpt-5.5-2026-04-23", "gpt-5.6-luna", "gpt-5.6-sol", "gpt-6-astra"], +) +def test_models_that_refuse_zero_get_their_only_supported_value(model: str) -> None: + assert resolve_temperature(model) == FIXED_TEMPERATURE == 1.0 + + +@pytest.mark.parametrize("model", ["gpt-4o-mini", "gpt-4.1", "gpt-5", "gpt-5.4-mini"]) +def test_every_other_model_still_gets_zero(model: str) -> None: + """Determinism stays the default; only the models that refuse it lose it. + + gpt-5 and gpt-5.4-mini are in this list deliberately: they are newer than + gpt-4o-mini and they do accept 0.0, so the rule is not "new models". + """ + assert resolve_temperature(model) == 0.0 + + +def test_the_prefix_table_covers_dated_snapshots() -> None: + """gpt-5.6-luna and a dated pin of it must resolve the same way. + + Matching on prefix is why this table needs no edit each time OpenAI pins a + snapshot of a family already listed. + """ + assert resolve_temperature("gpt-5.6-luna-2026-06-01") == FIXED_TEMPERATURE + + +def test_the_override_wins_over_the_table(monkeypatch: pytest.MonkeyPatch) -> None: + """The escape hatch, for a model the table has not met.""" + monkeypatch.setenv("LLM_TEMPERATURE", "1") + assert resolve_temperature("gpt-4o-mini") == 1.0 + monkeypatch.setenv("LLM_TEMPERATURE", "0") + assert resolve_temperature("gpt-5.6-luna") == 0.0 + + +@pytest.mark.parametrize("value", ["", " "]) +def test_an_empty_override_is_treated_as_unset( + value: str, monkeypatch: pytest.MonkeyPatch +) -> None: + """An empty env var in a compose file means "not configured", not 0.0.""" + monkeypatch.setenv("LLM_TEMPERATURE", value) + assert resolve_temperature("gpt-4o-mini") == 0.0 + + +def test_an_unparseable_override_stops_the_process( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Article IV: a temperature nobody can honour must not fall back silently.""" + monkeypatch.setenv("LLM_TEMPERATURE", "warm") + with pytest.raises(SystemExit, match="not a number"): + resolve_temperature("gpt-4o-mini") + + +def test_get_llm_sends_the_temperature_it_is_given( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setenv("OPENAI_API_KEY", "sk-not-a-real-key") # construction only + assert _built("gpt-4o-mini", temperature=0.0).temperature == 0.0 + assert _built("gpt-5.6-luna", temperature=1.0).temperature == 1.0 + + +def test_get_llm_still_defaults_to_zero(monkeypatch: pytest.MonkeyPatch) -> None: + """Callers that never heard of this change keep the old behaviour.""" + monkeypatch.setenv("OPENAI_API_KEY", "sk-not-a-real-key") + assert _built("gpt-4o-mini").temperature == 0.0