Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 19 additions & 2 deletions providers/asione.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,22 @@

logger = get_logger(__name__)

# Share of max_tokens reserved for reasoning at each effort level; the rest stays for the answer.
# Ratios follow OpenRouter: https://openrouter.ai/docs/guides/best-practices/reasoning-tokens#reasoning-effort-level
REASONING_EFFORT_RATIO = {
"none": 0.0,
"minimal": 0.10,
"low": 0.20,
"medium": 0.50,
"high": 0.80,
"xhigh": 0.95,
"max": 0.95,
}

def _reasoning_budget(max_tokens: int, effort: str) -> int:
"""Tokens reserved for reasoning; the rest of max_tokens stays for the answer."""
return int(max_tokens * REASONING_EFFORT_RATIO.get(str(effort).lower(), 0.0))

class ASIOneProvider(providers.LLMProvider):

def __init__(self):
Expand All @@ -31,8 +47,9 @@ class ASIOneProviderImpl(llm.AIProvider):

def convert_request(self, request: providers.LLMRequest) -> dict[str, Any]:
result = super().convert_request(request)
thinking_budget = _reasoning_budget(request.max_tokens, request.reasoning_mode)
result["extra_body"] = {
"enable_thinking": True,
"thinking_budget": 6000
"enable_thinking": thinking_budget > 0,
"thinking_budget": thinking_budget
}
return result
87 changes: 83 additions & 4 deletions providers/lib_llm_ext.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,18 +3,88 @@
from providers import *
from typing import Optional, Tuple, Dict, Any
from config import config_get_by_key
from src.logger import get_logger
import json
import uuid

PROMPT_DELIMITER = ":-:-:-:"

from src.logger import get_logger
LLM_EMPTY_RESPONSE_MESSAGE = (
"The agent didn\'t return an answer: reasoning exceeded the token limit for "
"this response before it could produce one."
"\n\n"
"If you are not an administrator: ask the Omega administrator to lower "
"the reasoning level or raise the response token budget - or try breaking "
"your request into smaller, simpler steps."
"\n\n"
"If you are the Omega administrator: check whether the model supports a "
"lower reasoning level and set it via 'reasoningMode' "
"(e.g. high → medium → low). Alternatively, raise 'maxOutputToken' - "
"reasoning and the final answer draw from the same token limit, so higher "
"reasoning levels need a higher token limit."
)
LLM_TRUNCATED_CALL_HINT = (
"The call was cut off by the output token limit. "
"Retry with shorter arguments or split the content into several calls."
)


logger = get_logger(__name__)

def _log_raw(kind, provider: str, model: str, raw: Dict) -> None:
logger.debug(f"[{kind}] provider={provider} model={model} raw={raw!r}")

def _log_chat_completion(provider: str, model: str, response) -> None:
"""Report how the completion budget was actually spent (Chat Completions API)."""
finish_reason = getattr(response.choices[0], "finish_reason", None)
usage = getattr(response, "usage", None)
details = getattr(usage, "completion_tokens_details", None)
prompt_details = getattr(usage, "prompt_tokens_details", None)
line = (
f"[LLM_USAGE] provider={provider} model={model} "
f"finish_reason={finish_reason} "
f"prompt_tokens={getattr(usage, 'prompt_tokens', None)} "
f"cached_tokens={getattr(prompt_details, 'cached_tokens', None)} "
f"completion_tokens={getattr(usage, 'completion_tokens', None)} "
f"reasoning_tokens={getattr(details, 'reasoning_tokens', None)} "
)
logger.info(line)

def _log_responses_completion(provider: str, model: str, response) -> None:
"""Report how the completion budget was actually spent (Responses API)."""
incomplete_details = getattr(response, "incomplete_details", None)
usage = getattr(response, "usage", None)
input_details = getattr(usage, "input_tokens_details", None)
output_details = getattr(usage, "output_tokens_details", None)
line = (
f"[LLM_USAGE] provider={provider} model={model} "
f"status={getattr(response, 'status', None)} "
f"incomplete_reason={getattr(incomplete_details, 'reason', None)} "
f"input_tokens={getattr(usage, 'input_tokens', None)} "
f"cached_tokens={getattr(input_details, 'cached_tokens', None)} "
f"output_tokens={getattr(usage, 'output_tokens', None)} "
f"reasoning_tokens={getattr(output_details, 'reasoning_tokens', None)} "
)
logger.info(line)

def _llm_empty_response_call(response_id: Optional[str]) -> LLMToolCall:
"""Build a `send` tool call that explains an empty LLM reply to the user.

Used when the LLM spends the entire output token budget on reasoning
and returns no tool calls, so there is no tool call id to reuse.

Args:
response_id: Id of the LLM response, used as the tool call id when present.

Returns:
Tool call that sends LLM_EMPTY_RESPONSE_MESSAGE.
"""
return (
LLMToolCall() \
.with_name("send") \
.with_id(response_id or f"call_{uuid.uuid4().hex}") \
.with_arguments({"content": LLM_EMPTY_RESPONSE_MESSAGE})
)

def _split_system_user(content: str) -> Tuple[str, str]:
"""
MeTTa sends:
Expand Down Expand Up @@ -146,18 +216,27 @@ def convert_request(self, request: LLMRequest) -> Dict[str, Any]:
}

def convert_response(self, raw):
_log_chat_completion(self._name, self._model_name, raw)
response = LLMResponse()

message = raw.choices[0].message
choice = raw.choices[0]
message = choice.message
exhausted = choice.finish_reason == "length"
if not message.tool_calls:
logger.warning("LLM returned an empty response")
if exhausted:
response.add_tool_call(_llm_empty_response_call(raw.id))
return response

for tool_call in message.tool_calls:
tc = LLMToolCall().with_name(tool_call.function.name).with_id(tool_call.id)
try:
arguments = json.loads(tool_call.function.arguments)
except json.JSONDecodeError as error:
response.add_tool_call(tc.with_error(f"Invalid tool arguments from model: {error}"))
error_text = f"Invalid tool arguments from model: {error}"
if exhausted:
error_text = f"{error_text}. {LLM_TRUNCATED_CALL_HINT}"
response.add_tool_call(tc.with_error(error_text))
else:
if isinstance(arguments, dict):
response.add_tool_call(tc.with_arguments(arguments))
Expand Down
34 changes: 12 additions & 22 deletions providers/openai.py
Original file line number Diff line number Diff line change
Expand Up @@ -90,44 +90,34 @@ def convert_request(self, request: LLMRequest) -> dict[str, Any]:

return result

def log_usage_statistics(self, response):
usage = getattr(response, "usage", None)
if usage:
input_tokens = getattr(usage, "input_tokens", None)
output_tokens = getattr(usage, "output_tokens", None)
total_tokens = getattr(usage, "total_tokens", None)
details = getattr(usage, "input_tokens_details", None)
cached_tokens = getattr(details, "cached_tokens", None) if details else None

logger.info(
f"[LLM_USAGE] provider={self._name} model={self._model_name} "
f"input_tokens={input_tokens} output_tokens={output_tokens} "
f"total_tokens={total_tokens} cached_tokens={cached_tokens}"
)

def convert_response(self, raw):
self.log_usage_statistics(raw)
llm._log_responses_completion(self._name, self._model_name, raw)

response = LLMResponse()
output = raw.output
if not output:
return response

for item in output:
exhausted = getattr(raw.incomplete_details, "reason", None) == "max_output_tokens"
for item in raw.output or []:
if item.type != "function_call":
continue
tool_call = item
tc = LLMToolCall().with_name(tool_call.name).with_id(tool_call.id)
try:
arguments = json.loads(tool_call.arguments)
except json.JSONDecodeError as error:
response.add_tool_call(tc.with_error(f"Invalid tool arguments from model: {error}"))
error_text = f"Invalid tool arguments from model: {error}"
if exhausted:
error_text = f"{error_text}. {llm.LLM_TRUNCATED_CALL_HINT}"
response.add_tool_call(tc.with_error(error_text))
else:
if isinstance(arguments, dict):
response.add_tool_call(tc.with_arguments(arguments))
else:
response.add_tool_call(tc.with_error("Tool arguments must be a JSON object"))

if not response.calls:
logger.warning("LLM returned an empty response")
if exhausted:
response.add_tool_call(llm._llm_empty_response_call(raw.id))

return response

def chat(self, request: LLMRequest) -> LLMResponse:
Expand Down
4 changes: 2 additions & 2 deletions providers/openrouter.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,8 +34,8 @@ def _openrouter_extra_body(self, request: providers.LLMRequest) -> dict[str, Any
sysmsg = request.messages[0].content
body = {
"reasoning": {
"enabled": True,
"max_tokens": request.max_tokens,
"enabled": True if request.reasoning_mode and str(request.reasoning_mode).lower() != "none" else False,
"effort": request.reasoning_mode,
"exclude": True,
}
}
Expand Down
Loading