diff --git a/.env.example b/.env.example index 38fdfee..68ee921 100644 --- a/.env.example +++ b/.env.example @@ -67,6 +67,12 @@ LLM_MODEL_TOOL= # LLM_MODEL_TOOL. Leave empty to disable escalation. Also settable at runtime via # the /model API (model_deep) or the /deep Telegram command. # Example: LLM_MODEL=claude-sonnet-5 with LLM_MODEL_DEEP=claude-opus-4-8 +# OpenAI Astra with high reasoning effort (API key or ChatGPT OAuth): +# LLM_MODEL_DEEP=gpt-6-astra-high +# With an Anthropic primary model, select OpenAI explicitly: +# LLM_MODEL_DEEP=openai/gpt-6-astra-high +# The -high suffix is a Lethe option; requests use model=gpt-6-astra and +# reasoning.effort=high. Requires OpenAI authentication alongside the main model. LLM_MODEL_DEEP= # Explicit provider override (anthropic | openai | openrouter | opencode-go). Usually diff --git a/CHANGELOG.md b/CHANGELOG.md index d54d9c6..4f4f6de 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,11 @@ ## 0.28.0 - Agent-id installs itself, browser CLI health probe +- **Deep thinking supports GPT-6 Astra with high reasoning effort.** Configure + `LLM_MODEL_DEEP=openai/gpt-6-astra-high` to keep the primary model while + escalating difficult work to Astra. API and ChatGPT subscription requests + use Responses with the bare model ID and `reasoning.effort=high`. + - **Scheduled wake delivery now closes cleanly after a confirmed Telegram result.** The model-facing Telegram contract now matches `/wake`: a normal final response is delivered automatically when no Telegram tool message was sent, while a diff --git a/README.md b/README.md index 5b9f7df..2186029 100644 --- a/README.md +++ b/README.md @@ -328,6 +328,8 @@ Lethe routes chat through `genai`. The runtime supports both API-key and subscri Lethe uses up to four model slots. `LLM_MODEL` is the main model; `LLM_MODEL_AUX` (defaults to the main model) handles lightweight/background calls (summarizer, curator, heartbeat). Two optional tiers let a turn change models mid-flight: `LLM_MODEL_TOOL` is a stronger reasoner a turn switches to the moment a tool is used, and `LLM_MODEL_DEEP` is a powerful "deep thinking" model the agent **escalates to on demand** for hard tasks — by calling the `think_deeply` tool (self-recognition), automatically when a turn is visibly struggling, or for a subagent spawned on the `deep` tier. Both reset to `LLM_MODEL` on the next turn; deep escalation outranks the tool switch. The deep tier can also be changed at runtime via `POST /model` (`model_deep`) or Telegram `/deep `; the tool tier is environment/config only. +For [GPT-6 Astra](https://developers.openai.com/api/docs/models/gpt-6-astra) with high reasoning effort, set `LLM_MODEL_DEEP=gpt-6-astra-high` in the runtime environment, or use `/deep gpt-6-astra-high` over Telegram for the current process. This requires OpenAI authentication (an API key or ChatGPT login) and the native OpenAI endpoint. When keeping an Anthropic primary model, use `LLM_MODEL_DEEP=openai/gpt-6-astra-high` to select OpenAI explicitly. The `-high` suffix is Lethe's configuration shorthand: requests use `model: "gpt-6-astra"` and `reasoning: {"effort": "high"}` through the Responses API. Astra retains Lethe's 128k compaction budget. + ### Subscription OAuth `lethe login openai` runs a device-code flow against `auth.openai.com`; tokens land in `~/.lethe/credentials/openai_oauth_tokens.json`. Calls then go to the Codex Responses API at `chatgpt.com/backend-api/codex/responses` using your ChatGPT Plus/Pro session — no `OPENAI_API_KEY` needed. Override the token file with `LETHE_OPENAI_OAUTH_TOKENS` or supply a raw token via `OPENAI_AUTH_TOKEN`. diff --git a/config/model_catalog.json b/config/model_catalog.json index fce78de..07e4e74 100644 --- a/config/model_catalog.json +++ b/config/model_catalog.json @@ -1,5 +1,5 @@ { - "_updated": "2026-05-29", + "_updated": "2026-09-17", "_note": "Curated model catalog for /model and /aux commands. Only recent models (≤5mo), tool-capable, ≥100k ctx. First entry per list is the wizard default.", "openrouter": { "main": [ @@ -39,6 +39,8 @@ "openai": { "main": [ ["GPT-5.5", "gpt-5.5", "$5/$30"], + ["GPT-6 Astra", "gpt-6-astra", "$10/$50"], + ["GPT-6 Astra (high reasoning)", "gpt-6-astra-high", "$10/$50"], ["GPT-5.5 Pro", "gpt-5.5-pro", "$30/$180"], ["GPT-5.4", "gpt-5.4", "$2.50/$15"], ["GPT-5.4 Pro", "gpt-5.4-pro", "$5/$30"], diff --git a/config/model_context_limits.json b/config/model_context_limits.json index 6406dc4..743836c 100644 --- a/config/model_context_limits.json +++ b/config/model_context_limits.json @@ -1,6 +1,6 @@ { "_note": "Per-model context window in tokens, used to auto-derive the compaction budget. Capped at 128k by policy — even when a model technically supports more (Gemini 1M, Claude 200k, GPT-5 400k), running closer to the cap is slow, costs linearly more on input, and degrades attention. The agent's auto-compaction + summarizer keep history well below 128k anyway, so nothing here should exceed 128000. Falls back to LLM_CONTEXT_LIMIT (default 100k) when a model id isn't listed. Override per-deployment via LLM_CONTEXT_LIMIT.", - "_updated": "2026-05-31", + "_updated": "2026-09-17", "claude-opus-4-8": 128000, "claude-opus-4-7": 128000, @@ -17,6 +17,10 @@ "openrouter/anthropic/claude-haiku-4.5": 128000, "gpt-5.6-terra": 128000, + "gpt-6-astra": 128000, + "gpt-6-astra-high": 128000, + "openai/gpt-6-astra": 128000, + "openai/gpt-6-astra-high": 128000, "gpt-5.5": 128000, "gpt-5": 128000, "gpt-5.4": 128000, diff --git a/src/llm/client.rs b/src/llm/client.rs index 90fd11f..b574e3b 100644 --- a/src/llm/client.rs +++ b/src/llm/client.rs @@ -311,7 +311,7 @@ fn rejects_sampling_params(model: &str) -> bool { if name.contains('/') { return false; } - is_gpt5_reasoning(name) || is_o_series(name) + is_gpt5_reasoning(name) || name.starts_with("gpt-6-astra") || is_o_series(name) } #[derive(Clone)] @@ -1256,7 +1256,7 @@ fn slash_provider(model: &str) -> Option<&str> { } } -fn strip_slash_provider<'a>(model: &'a str, provider: &str) -> &'a str { +pub(super) fn strip_slash_provider<'a>(model: &'a str, provider: &str) -> &'a str { let Some((prefix, rest)) = model.split_once('/') else { return model; }; @@ -1270,18 +1270,20 @@ fn strip_slash_provider<'a>(model: &'a str, provider: &str) -> &'a str { /// Adapter for `provider`, refined by the model where the protocol depends on /// it. /// -/// OpenAI's gpt-5 reasoning family cannot combine function tools with reasoning -/// on `/v1/chat/completions` — the API rejects the request outright and points +/// OpenAI's GPT-5 reasoning family and GPT-6 Astra cannot combine function tools +/// with reasoning on `/v1/chat/completions` — the API rejects the request and points /// at `/v1/responses`, which supports both. An agent request always carries -/// tools, so those models go to the Responses adapter; otherwise every gpt-5.x -/// turn would have to give up reasoning to keep its tools. +/// tools, so those models go to the Responses adapter. /// /// Only the direct `openai` provider is upgraded. OpenRouter also speaks the /// OpenAI protocol but has no `/v1/responses`, and it has its own branch in /// [`router_target_for_model`] that pins `AdapterKind::OpenAI`. fn adapter_for(provider: &str, model_name: &str) -> Option { let adapter = adapter_for_provider(provider)?; - if provider == "openai" && adapter == AdapterKind::OpenAI && is_gpt5_reasoning(model_name) { + if provider == "openai" + && adapter == AdapterKind::OpenAI + && (is_gpt5_reasoning(model_name) || model_name.starts_with("gpt-6-astra")) + { return Some(AdapterKind::OpenAIResp); } Some(adapter) @@ -1395,10 +1397,11 @@ fn should_use_anthropic_oauth(model: &str, config: &LlmRouterConfig) -> bool { if normalize_api_base(&config.api_base).is_some() { return false; } - if slash_provider(model) == Some("openrouter") { - return false; - } - if slash_provider(model) == Some("opencode-go") { + // Explicit cross-provider tiers must not inherit the primary model's OAuth. + if matches!( + slash_provider(model), + Some("openai" | "openrouter" | "opencode-go") + ) { return false; } if normalized_provider(&config.provider).as_deref() == Some("openrouter") { @@ -1423,6 +1426,10 @@ fn should_use_openai_oauth(model: &str, config: &LlmRouterConfig) -> bool { if slash_provider(model) == Some("opencode-go") { return false; } + // Explicit tier providers take precedence over the primary model's provider. + if slash_provider(model) == Some("openai") { + return true; + } if normalized_provider(&config.provider).as_deref() == Some("openrouter") { return false; } @@ -1435,8 +1442,7 @@ fn should_use_openai_oauth(model: &str, config: &LlmRouterConfig) -> bool { // still work — having an OAuth token doesn't override an explicit // OPENAI_API_KEY for openrouter or custom api_base targets, since // those branches return false above. - slash_provider(model) == Some("openai") - || normalized_provider(&config.provider).as_deref() == Some("openai") + normalized_provider(&config.provider).as_deref() == Some("openai") } pub fn llm_auth_mode_for_settings(settings: &Settings) -> String { @@ -2936,13 +2942,15 @@ mod tests { } #[test] - fn direct_openai_gpt5_routes_to_the_responses_adapter() { + fn direct_openai_reasoning_models_route_to_the_responses_adapter() { // gpt-5 reasoning models reject function tools on /v1/chat/completions // and must go to /v1/responses, which supports tools + reasoning. for model in [ "openai/gpt-5.6-terra", "openai/gpt-5.5", "openai/gpt-5.4-mini", + "openai/gpt-6-astra", + "openai/gpt-6-astra-high", ] { let config = config_for(model, ""); let target = router_target_for_model(model, &config).unwrap(); @@ -2987,6 +2995,9 @@ mod tests { "gpt-5.5", "gpt-5.4-mini", "gpt-5", + "gpt-6-astra", + "gpt-6-astra-high", + "openai/gpt-6-astra-high", "o3-mini", "o1", ] { @@ -3007,6 +3018,7 @@ mod tests { "gpt-4o", // Relayed ids are normalized by the relay, so they are left alone. "openrouter/openai/gpt-5.4", + "openrouter/openai/gpt-6-astra-high", "openrouter/anthropic/claude-opus-4.7", ] { assert_eq!( @@ -3098,6 +3110,52 @@ mod tests { assert!(should_use_openai_oauth(&config.model, &config)); } + #[test] + fn openai_deep_model_does_not_inherit_anthropic_oauth() { + let config = config_for("claude-opus-4-8", "anthropic"); + assert!(should_use_anthropic_oauth(&config.model, &config)); + assert!(!should_use_openai_oauth(&config.model, &config)); + + for model in ["openai/gpt-6-astra-high", "OpenAI/gpt-6-astra-high"] { + assert!(should_use_openai_oauth(model, &config)); + // Without OpenAI OAuth, execution must continue to the API-key client. + assert!(!should_use_anthropic_oauth(model, &config)); + let target = router_target_for_model(model, &config).unwrap(); + assert_eq!(target.auth_env, "OPENAI_API_KEY"); + assert_eq!(target.endpoint, OPENAI_ENDPOINT); + assert_eq!(target.adapter, AdapterKind::OpenAIResp); + assert_eq!(target.model_name, "gpt-6-astra-high"); + } + } + + #[test] + fn openai_deep_model_overrides_relay_provider_for_oauth() { + for provider in ["openrouter", "opencode-go"] { + let mut config = config_for("relay-model", provider); + assert!(!should_use_openai_oauth(&config.model, &config)); + + for model in ["openai/gpt-6-astra-high", "OpenAI/gpt-6-astra-high"] { + assert!(should_use_openai_oauth(model, &config)); + assert!(!should_use_anthropic_oauth(model, &config)); + let target = router_target_for_model(model, &config).unwrap(); + assert_eq!(target.auth_env, "OPENAI_API_KEY"); + assert_eq!(target.endpoint, OPENAI_ENDPOINT); + assert_eq!(target.adapter, AdapterKind::OpenAIResp); + } + + config.provider = "openai".to_string(); + let relayed_model = format!("{provider}/openai/gpt-6-astra-high"); + assert!(!should_use_openai_oauth(&relayed_model, &config)); + + config.api_base = "http://localhost:8080/v1/".to_string(); + let model = "openai/gpt-6-astra-high"; + assert!(!should_use_openai_oauth(model, &config)); + let target = router_target_for_model(model, &config).unwrap(); + assert_eq!(target.auth_env, "OPENAI_API_KEY"); + assert_eq!(target.endpoint, config.api_base); + } + } + #[test] fn openai_oauth_skips_openrouter_and_custom_api_base() { // openrouter slash-prefix wins regardless of provider diff --git a/src/llm/openai_oauth.rs b/src/llm/openai_oauth.rs index d133874..a019f19 100644 --- a/src/llm/openai_oauth.rs +++ b/src/llm/openai_oauth.rs @@ -25,7 +25,7 @@ use futures_util::StreamExt; use genai::adapter::AdapterKind; use genai::chat::{ ChatOptions, ChatRequest, ChatResponse, ChatRole, ContentPart, MessageContent, - PromptTokensDetails, ToolCall, Usage, + PromptTokensDetails, ReasoningEffort, ToolCall, Usage, }; use genai::{ModelIden, chat::Tool}; use reqwest::StatusCode; @@ -594,7 +594,7 @@ fn trim_openai_input_items(body: &mut Value) { } } -fn openai_responses_body(model: &str, request: ChatRequest, _options: &ChatOptions) -> Value { +fn openai_responses_body(model: &str, request: ChatRequest, options: &ChatOptions) -> Value { // System → instructions; everything else → typed input items. // max_tokens is intentionally not forwarded — the Codex endpoint // rejects token-limit params. @@ -636,6 +636,8 @@ fn openai_responses_body(model: &str, request: ChatRequest, _options: &ChatOptio instructions_parts.join("\n\n") }; + let model = super::client::strip_slash_provider(model, "openai"); + let (model_effort, model) = ReasoningEffort::from_model_name(model); let mut body = json!({ "model": model, "instructions": instructions, @@ -643,6 +645,14 @@ fn openai_responses_body(model: &str, request: ChatRequest, _options: &ChatOptio "store": false, "stream": true, }); + if let Some(effort) = options + .reasoning_effort + .as_ref() + .or(model_effort.as_ref()) + .and_then(ReasoningEffort::as_keyword) + { + body["reasoning"] = json!({"effort": effort}); + } trim_openai_input_items(&mut body); if let Some(tools) = request.tools && !tools.is_empty() @@ -1682,6 +1692,8 @@ mod tests { #[test] fn body_extracts_instructions_and_input_items() { let body = openai_responses_body("gpt-5.2", make_request(), &ChatOptions::default()); + assert_eq!(body["model"], json!("gpt-5.2")); + assert!(body.get("reasoning").is_none()); assert_eq!(body["instructions"], json!("be precise")); assert_eq!(body["store"], json!(false)); assert_eq!(body["stream"], json!(true)); @@ -1701,6 +1713,38 @@ mod tests { assert!(tools[0]["parameters"].is_object()); } + #[test] + fn astra_high_body_sends_reasoning_effort_with_function_tools() { + let options = ChatOptions::default() + .with_temperature(0.7) + .with_top_p(0.9) + .with_max_tokens(2048); + for model in [ + "gpt-6-astra-high", + "openai/gpt-6-astra-high", + "OpenAI/gpt-6-astra-high", + ] { + let body = openai_responses_body(model, make_request(), &options); + + assert_eq!(body["model"], json!("gpt-6-astra")); + assert_eq!(body["reasoning"], json!({"effort": "high"})); + assert_eq!(body["tools"][0]["type"], json!("function")); + assert_eq!(body["tools"][0]["name"], json!("lookup")); + for parameter in ["temperature", "top_p", "top_logprobs", "max_tokens"] { + assert!(body.get(parameter).is_none(), "unexpected {parameter}"); + } + } + } + + #[test] + fn explicit_reasoning_effort_overrides_model_suffix() { + let options = ChatOptions::default().with_reasoning_effort(ReasoningEffort::Medium); + let body = openai_responses_body("gpt-6-astra-high", make_request(), &options); + + assert_eq!(body["model"], json!("gpt-6-astra")); + assert_eq!(body["reasoning"], json!({"effort": "medium"})); + } + #[test] fn body_trims_oldest_input_items_when_too_large() { let huge = "x".repeat(30_000); diff --git a/vendor/genai/LETHE_FORK.md b/vendor/genai/LETHE_FORK.md index 47ae4b1..7e6c0f7 100644 --- a/vendor/genai/LETHE_FORK.md +++ b/vendor/genai/LETHE_FORK.md @@ -59,8 +59,8 @@ Prompt-caching changes: Anthropic's, so both paths must agree on this mapping by construction rather than through a copy that can drift. - `src/adapter/adapters/openai_resp/adapter_impl.rs`: one added test, - `lethe_fork_agent_turn_on_a_gpt5_reasoning_model`. No production code touched. - It pins the request shape a Lethe agent turn produces for gpt-5 — the URL, + `lethe_fork_agent_turn_on_a_reasoning_model`. No production code touched. + It pins the request shape a Lethe agent turn produces for gpt-5 and Astra — the URL, `max_output_tokens`, tools surviving, and no `temperature` — since that route is the whole reason those models are sent to the Responses API. - `Cargo.toml`: the published crate's `[[example]]`/`[[test]]` target @@ -133,6 +133,15 @@ recorded so nobody re-applies them: reasoning models to `AdapterKind::OpenAIResp` (see `adapter_for()` in `src/llm/client.rs`). `/v1/responses` supports tools *and* reasoning together. +### GPT-6 Astra + +Direct OpenAI Astra requests use the Responses API for tool support, including +streaming. Lethe's model-aware request options omit unsupported sampling +parameters for Astra. Reasoning suffixes such as +`gpt-6-astra-high` select the bare model with `reasoning.effort=high`. +Lethe's ChatGPT OAuth transport applies the same shorthand separately and +accepts an explicit `openai/` prefix for mixed-provider model tiers. + ## Tracking upstream The remaining patches are genuine upstream gaps, not workarounds. The diff --git a/vendor/genai/src/adapter/adapter_kind.rs b/vendor/genai/src/adapter/adapter_kind.rs index e6ac8ce..fac7cb1 100644 --- a/vendor/genai/src/adapter/adapter_kind.rs +++ b/vendor/genai/src/adapter/adapter_kind.rs @@ -321,6 +321,7 @@ impl AdapterKind { // migh be a little generic on this one { if model.starts_with("gpt-5") + || model.starts_with("gpt-6-astra") || (model.starts_with("gpt") && (model.contains("codex") || model.contains("pro"))) { Ok(Self::OpenAIResp) @@ -382,3 +383,19 @@ impl AdapterKind { } // endregion: --- Support + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn astra_uses_responses_unless_adapter_is_explicit() { + for model in ["gpt-6-astra", "gpt-6-astra-high", "gpt-5.6-sol"] { + assert_eq!(AdapterKind::from_model(model).unwrap(), AdapterKind::OpenAIResp); + } + assert_eq!( + AdapterKind::from_model("openai::gpt-6-astra-high").unwrap(), + AdapterKind::OpenAI + ); + } +} diff --git a/vendor/genai/src/adapter/adapters/openai_resp/adapter_impl.rs b/vendor/genai/src/adapter/adapters/openai_resp/adapter_impl.rs index 98c123f..8fbcff8 100644 --- a/vendor/genai/src/adapter/adapters/openai_resp/adapter_impl.rs +++ b/vendor/genai/src/adapter/adapters/openai_resp/adapter_impl.rs @@ -767,51 +767,41 @@ mod tests { ); } - /// Lethe fork: the exact request shape a Lethe agent turn produces for a - /// gpt-5 reasoning model, which is why those models are routed here at all. - /// - /// On /v1/chat/completions this same request is a hard 400 — "Function tools - /// with reasoning_effort are not supported for gpt-5.6-terra ... use - /// /v1/responses or set reasoning_effort to 'none'" — because an agent - /// always carries tools and the model's implicit default effort is not - /// "none". The Responses API supports tools and reasoning together, so no - /// effort has to be forced here. - /// - /// Guards the three things that would silently break the route: the URL, - /// `max_output_tokens` (not `max_tokens`), and the absence of `temperature`, - /// which reasoning models reject and which Lethe drops in `chat_options()`. + /// Match Lethe's request options: sampling omitted, usage captured when streaming. #[test] - fn lethe_fork_agent_turn_on_a_gpt5_reasoning_model() { - let chat_options = ChatOptions::default().with_max_tokens(64); - let options_set = ChatOptionsSet::default().with_chat_options(Some(&chat_options)); - let target = ServiceTarget { - model: ModelIden::new(AdapterKind::OpenAIResp, "gpt-5.6-terra"), - auth: AuthData::from_single("test-key"), - endpoint: OpenAIRespAdapter::default_endpoint(), - }; - let chat_req = ChatRequest::from_user("hi").append_tool(Tool::new("get_time")); - - let web_req = OpenAIRespAdapter::to_web_request_data(target, ServiceType::ChatStream, chat_req, options_set) - .expect("a gpt-5 agent turn must build a Responses request"); + fn lethe_fork_agent_turn_on_a_reasoning_model() { + for model_name in ["gpt-5.6-terra", "gpt-6-astra", "gpt-6-astra-high"] { + for service_type in [ServiceType::Chat, ServiceType::ChatStream] { + let chat_options = ChatOptions::default().with_max_tokens(64).with_capture_usage(true); + let options_set = ChatOptionsSet::default().with_chat_options(Some(&chat_options)); + let target = ServiceTarget { + model: ModelIden::new(AdapterKind::OpenAIResp, model_name), + auth: AuthData::from_single("test-key"), + endpoint: OpenAIRespAdapter::default_endpoint(), + }; + let chat_req = ChatRequest::from_user("hi").append_tool(Tool::new("get_time")); + let web_req = OpenAIRespAdapter::to_web_request_data(target, service_type, chat_req, options_set) + .expect("an agent turn must build a Responses request"); - assert!( - web_req.url.ends_with("/responses"), - "must target /v1/responses, got {}", - web_req.url - ); - assert_eq!(web_req.payload["model"], json!("gpt-5.6-terra")); - assert_eq!(web_req.payload["stream"], json!(true)); - assert!(web_req.payload["tools"].is_array(), "tools must survive"); - assert_eq!( - web_req.payload["max_output_tokens"], - json!(64), - "the Responses API takes max_output_tokens, not max_tokens" - ); - assert!(web_req.payload.get("max_tokens").is_none()); - assert!( - web_req.payload.get("temperature").is_none(), - "reasoning models reject temperature; Lethe must not send one" - ); + assert_eq!(web_req.url, "https://api.openai.com/v1/responses"); + if matches!(service_type, ServiceType::ChatStream) { + assert_eq!(web_req.payload["stream"], true); + } + assert_eq!(web_req.payload["tools"][0]["type"], "function"); + assert_eq!(web_req.payload["tools"][0]["name"], "get_time"); + assert_eq!(web_req.payload["max_output_tokens"], 64); + for parameter in ["max_tokens", "temperature", "top_p", "top_logprobs", "stream_options"] { + assert!(web_req.payload.get(parameter).is_none(), "unexpected {parameter}"); + } + if model_name.ends_with("-high") { + assert_eq!(web_req.payload["model"], "gpt-6-astra"); + assert_eq!(web_req.payload["reasoning"]["effort"], "high"); + } else { + assert_eq!(web_req.payload["model"], model_name); + assert!(web_req.payload.get("reasoning").is_none()); + } + } + } } }