Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions deploy/beta/env.beta.template
Original file line number Diff line number Diff line change
Expand Up @@ -22,4 +22,12 @@ CHAINLIT_URL=https://beta.reactome.org
# anything else makes retrieval meaningless.
#EMBEDDING_MODEL=
# LLM used for answers; defaults to gpt-4o-mini.
# gpt-5.6-luna is verified to work end to end. It answers with noticeably more
# grounding in the retrieved Reactome records, and takes roughly twice as long
# per question (41s vs 22s, averaged over three questions).
#LLM_MODEL=
# Sampling temperature. Leave unset: it defaults to 0.0, except for the
# gpt-5.5/5.6/6 families, which accept only their own default of 1 and return a
# 400 for anything else. Set this only for a model that rule does not yet know
# about; see resolve_temperature in src/agent/graph.py.
#LLM_TEMPERATURE=
8 changes: 8 additions & 0 deletions env_template
Original file line number Diff line number Diff line change
Expand Up @@ -20,4 +20,12 @@ TAVILY_API_KEY=
# anything else makes retrieval meaningless.
#EMBEDDING_MODEL=
# LLM used for answers; defaults to gpt-4o-mini.
# gpt-5.6-luna is verified to work end to end. It answers with noticeably more
# grounding in the retrieved Reactome records, and takes roughly twice as long
# per question (41s vs 22s, averaged over three questions).
#LLM_MODEL=
# Sampling temperature. Leave unset: it defaults to 0.0, except for the
# gpt-5.5/5.6/6 families, which accept only their own default of 1 and return a
# 400 for anything else. Set this only for a model that rule does not yet know
# about; see resolve_temperature in src/agent/graph.py.
#LLM_TEMPERATURE=
42 changes: 41 additions & 1 deletion src/agent/graph.py
Original file line number Diff line number Diff line change
Expand Up @@ -78,6 +78,42 @@ def resolve_embedding_model() -> str:
return configured or bundle_model


# Models that accept only one temperature: their own default of 1. Any other
# value, 0.0 included, is a 400 on the first request rather than an error at
# construction:
#
# Unsupported value: 'temperature' does not support 0.0 with this model.
# Only the default (1) value is supported.
#
# Sending nothing is not an option -- ChatOpenAI supplies its own default of 0.7
# when the argument is omitted, and these models reject that too -- so the value
# has to be 1.0 explicitly.
#
# Unlike the embedding model, which is read from the bundle that built it, this
# cannot be derived: no endpoint reports which values a model accepts. Matched on
# prefix so dated snapshots (gpt-5.5-2026-04-23) and new members of a family
# need no edit. LLM_TEMPERATURE overrides, for a model this list has not met.
FIXED_TEMPERATURE_MODEL_PREFIXES = ("gpt-5.5", "gpt-5.6", "gpt-6")
FIXED_TEMPERATURE = 1.0


def resolve_temperature(model: str) -> float:
"""The temperature to send for `model`.

Returning 1.0 for the models above trades determinism for being able to use
them at all. That trade is made here, once, rather than at each call site.
"""
override = os.getenv("LLM_TEMPERATURE")
if override is not None and override.strip() != "":
try:
return float(override)
except ValueError:
raise SystemExit(f"LLM_TEMPERATURE={override!r} is not a number.") from None
if model.startswith(FIXED_TEMPERATURE_MODEL_PREFIXES):
return FIXED_TEMPERATURE
return 0.0


class AgentGraph:
def __init__(
self,
Expand All @@ -88,7 +124,11 @@ def __init__(
llm_model = os.getenv("LLM_MODEL", "gpt-4o-mini")
llm_base_url = os.getenv("LLM_BASE_URL", None)
llm: BaseChatModel = get_llm(
"openai", llm_model, base_url=llm_base_url, request_timeout=360.0
"openai",
llm_model,
base_url=llm_base_url,
request_timeout=360.0,
temperature=resolve_temperature(llm_model),
)
embedding_base_url = os.getenv("OPENAI_BASE_URL", None)
embedding: Embeddings = get_embedding(
Expand Down
17 changes: 15 additions & 2 deletions src/agent/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -49,20 +49,33 @@ def get_llm(
*,
base_url: str | None = None,
request_timeout: float | None = None,
temperature: float = 0.0,
) -> BaseChatModel:
"""Build a chat model. See `agent.graph.resolve_temperature` for the value.

0.0 is what this repository wants everywhere -- the graders, the intent
classifier and the query expander should all give the same answer twice.
Some models refuse it, which is why this is a parameter rather than the
constant it used to be.

There is no way to send *no* temperature on langchain-openai 0.2.14:
omitting the argument makes ChatOpenAI send its own pydantic default of 0.7,
which the models that refuse 0.0 refuse just as firmly. So the caller must
pass a value the model accepts; it cannot opt out.
"""
if model is None:
provider, model = provider.split("/", 1)
if provider == "openai":
return ChatOpenAI(
model=model,
temperature=0.0,
temperature=temperature,
base_url=base_url,
request_timeout=request_timeout, # type: ignore[call-arg] # pydantic-generated __init__
)
if provider == "ollama":
return ChatOllama(
model=model,
temperature=0.0,
temperature=temperature,
base_url=base_url,
request_timeout=request_timeout, # type: ignore[call-arg] # pydantic-generated __init__
)
Expand Down
10 changes: 9 additions & 1 deletion src/agent/tasks/completeness_grader.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,5 +28,13 @@ class CompletenessGrade(BaseModel):
)


# json_schema, not the default function_calling: the gpt-5.6 family refuses
# function tools on /v1/chat/completions ("Function tools with reasoning_effort
# are not supported ... use /v1/responses or set reasoning_effort to 'none'"),
# and langchain-openai 0.2.14 has no Responses API support. json_schema uses
# response_format instead, which every model here accepts -- verified against
# gpt-4o-mini and gpt-5.6-luna for all three graders.
def create_completeness_grader(llm: BaseChatModel) -> Runnable:
return completeness_prompt | llm.with_structured_output(CompletenessGrade)
return completeness_prompt | llm.with_structured_output(
CompletenessGrade, method="json_schema"
)
10 changes: 9 additions & 1 deletion src/agent/tasks/intent_classifier.py
Original file line number Diff line number Diff line change
Expand Up @@ -57,5 +57,13 @@ def resolve_active_sources(
return [next(iter(available_sources))]


# json_schema, not the default function_calling: the gpt-5.6 family refuses
# function tools on /v1/chat/completions ("Function tools with reasoning_effort
# are not supported ... use /v1/responses or set reasoning_effort to 'none'"),
# and langchain-openai 0.2.14 has no Responses API support. json_schema uses
# response_format instead, which every model here accepts -- verified against
# gpt-4o-mini and gpt-5.6-luna for all three graders.
def create_intent_classifier(llm: BaseChatModel) -> Runnable:
return intent_classifier_prompt | llm.with_structured_output(QueryIntent)
return intent_classifier_prompt | llm.with_structured_output(
QueryIntent, method="json_schema"
)
10 changes: 9 additions & 1 deletion src/agent/tasks/safety_checker.py
Original file line number Diff line number Diff line change
Expand Up @@ -69,5 +69,13 @@ class SafetyCheck(BaseModel):
)


# json_schema, not the default function_calling: the gpt-5.6 family refuses
# function tools on /v1/chat/completions ("Function tools with reasoning_effort
# are not supported ... use /v1/responses or set reasoning_effort to 'none'"),
# and langchain-openai 0.2.14 has no Responses API support. json_schema uses
# response_format instead, which every model here accepts -- verified against
# gpt-4o-mini and gpt-5.6-luna for all three graders.
def create_safety_checker(llm: BaseChatModel) -> Runnable:
return safety_check_prompt | llm.with_structured_output(SafetyCheck)
return safety_check_prompt | llm.with_structured_output(
SafetyCheck, method="json_schema"
)
102 changes: 102 additions & 0 deletions tests/agent/test_model_temperature.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,102 @@
"""Which temperature gets sent, and to which models.

The gpt-5.5/5.6 families and gpt-6-astra accept only their own default of 1.
Every other value is a 400 on the *first request*, not at construction, so
nothing catches it until a user asks a question.

There is no way to send no temperature at all: omitting the argument makes
ChatOpenAI send its own pydantic default of 0.7, which those models reject just
as firmly as 0.0. An earlier version of this change tried exactly that, and
`model_dump(exclude_unset=True)` agreed the field was unset -- while the request
still carried 0.7. So the assertions below are about the value chosen, and the
live behaviour was verified by hand against the API.
"""

import pytest

pytest.importorskip("langchain_openai", reason="LLM stack not installed")

from langchain_openai.chat_models.base import ChatOpenAI # noqa: E402

from agent.graph import FIXED_TEMPERATURE, resolve_temperature # noqa: E402
from agent.models import get_llm # noqa: E402


def _built(model: str, **kwargs: float) -> ChatOpenAI:
"""get_llm is typed BaseChatModel; the temperature lives on ChatOpenAI."""
llm = get_llm("openai", model, **kwargs) # type: ignore[arg-type]
assert isinstance(llm, ChatOpenAI)
return llm


@pytest.fixture(autouse=True)
def _no_override(monkeypatch: pytest.MonkeyPatch) -> None:
"""A developer's LLM_TEMPERATURE must not decide what these tests assert."""
monkeypatch.delenv("LLM_TEMPERATURE", raising=False)


@pytest.mark.parametrize(
"model",
["gpt-5.5", "gpt-5.5-2026-04-23", "gpt-5.6-luna", "gpt-5.6-sol", "gpt-6-astra"],
)
def test_models_that_refuse_zero_get_their_only_supported_value(model: str) -> None:
assert resolve_temperature(model) == FIXED_TEMPERATURE == 1.0


@pytest.mark.parametrize("model", ["gpt-4o-mini", "gpt-4.1", "gpt-5", "gpt-5.4-mini"])
def test_every_other_model_still_gets_zero(model: str) -> None:
"""Determinism stays the default; only the models that refuse it lose it.

gpt-5 and gpt-5.4-mini are in this list deliberately: they are newer than
gpt-4o-mini and they do accept 0.0, so the rule is not "new models".
"""
assert resolve_temperature(model) == 0.0


def test_the_prefix_table_covers_dated_snapshots() -> None:
"""gpt-5.6-luna and a dated pin of it must resolve the same way.

Matching on prefix is why this table needs no edit each time OpenAI pins a
snapshot of a family already listed.
"""
assert resolve_temperature("gpt-5.6-luna-2026-06-01") == FIXED_TEMPERATURE


def test_the_override_wins_over_the_table(monkeypatch: pytest.MonkeyPatch) -> None:
"""The escape hatch, for a model the table has not met."""
monkeypatch.setenv("LLM_TEMPERATURE", "1")
assert resolve_temperature("gpt-4o-mini") == 1.0
monkeypatch.setenv("LLM_TEMPERATURE", "0")
assert resolve_temperature("gpt-5.6-luna") == 0.0


@pytest.mark.parametrize("value", ["", " "])
def test_an_empty_override_is_treated_as_unset(
value: str, monkeypatch: pytest.MonkeyPatch
) -> None:
"""An empty env var in a compose file means "not configured", not 0.0."""
monkeypatch.setenv("LLM_TEMPERATURE", value)
assert resolve_temperature("gpt-4o-mini") == 0.0


def test_an_unparseable_override_stops_the_process(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""Article IV: a temperature nobody can honour must not fall back silently."""
monkeypatch.setenv("LLM_TEMPERATURE", "warm")
with pytest.raises(SystemExit, match="not a number"):
resolve_temperature("gpt-4o-mini")


def test_get_llm_sends_the_temperature_it_is_given(
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setenv("OPENAI_API_KEY", "sk-not-a-real-key") # construction only
assert _built("gpt-4o-mini", temperature=0.0).temperature == 0.0
assert _built("gpt-5.6-luna", temperature=1.0).temperature == 1.0


def test_get_llm_still_defaults_to_zero(monkeypatch: pytest.MonkeyPatch) -> None:
"""Callers that never heard of this change keep the old behaviour."""
monkeypatch.setenv("OPENAI_API_KEY", "sk-not-a-real-key")
assert _built("gpt-4o-mini").temperature == 0.0
Loading