From 68e3aef021504752a94315950632995ee55a904d Mon Sep 17 00:00:00 2001 From: CraftBot Date: Thu, 1 Oct 2026 11:21:20 +0900 Subject: [PATCH 1/5] Provide LLM reasoning effort param --- agent_core/core/impl/llm/cache/gemini.py | 17 +- agent_core/core/impl/llm/interface.py | 42 +- agent_core/core/impl/llm/reasoning_wire.py | 87 ++ .../impl/llm/transports/anthropic_messages.py | 28 +- .../impl/llm/transports/bedrock_converse.py | 18 + .../impl/llm/transports/chat_completions.py | 23 +- .../core/impl/llm/transports/gemini_native.py | 30 +- agent_core/core/llm/google_gemini_client.py | 46 +- .../models/chatgpt_subscription_client.py | 27 +- agent_core/core/models/factory.py | 15 +- agent_core/core/models/reasoning.py | 814 ++++++++++++++++++ scripts/probe_reasoning.py | 227 +++++ tests/llm/golden/conftest.py | 337 ++++++++ tests/llm/golden/snapshots/anthropic.json | 195 +++++ tests/llm/golden/snapshots/bedrock.json | 245 ++++++ tests/llm/golden/snapshots/byteplus.json | 83 ++ tests/llm/golden/snapshots/deepseek.json | 158 ++++ tests/llm/golden/snapshots/gemini.json | 242 ++++++ tests/llm/golden/snapshots/groq.json | 146 ++++ tests/llm/golden/snapshots/openai.json | 12 +- .../golden/snapshots/openrouter_claude.json | 181 ++++ .../snapshots/openrouter_non_claude.json | 158 ++++ tests/llm/golden/snapshots/remote.json | 72 ++ .../golden/snapshots/unruled_anthropic.json | 183 ++++ .../llm/golden/snapshots/unruled_bedrock.json | 220 +++++ .../llm/golden/snapshots/unruled_gemini.json | 230 +++++ tests/llm/golden/snapshots/unruled_grok.json | 158 ++++ .../llm/golden/snapshots/unruled_openai.json | 154 ++++ .../snapshots/unruled_openrouter_claude.json | 173 ++++ tests/llm/golden/test_golden_payloads.py | 103 ++- tests/llm/test_reasoning_rules.py | 449 ++++++++++ tests/llm/test_reliability.py | 244 ++++++ tests/test_bedrock_token_normalization.py | 3 + 33 files changed, 5069 insertions(+), 51 deletions(-) create mode 100644 agent_core/core/impl/llm/reasoning_wire.py create mode 100644 agent_core/core/models/reasoning.py create mode 100644 scripts/probe_reasoning.py create mode 100644 tests/llm/golden/conftest.py create mode 100644 tests/llm/golden/snapshots/anthropic.json create mode 100644 tests/llm/golden/snapshots/bedrock.json create mode 100644 tests/llm/golden/snapshots/byteplus.json create mode 100644 tests/llm/golden/snapshots/deepseek.json create mode 100644 tests/llm/golden/snapshots/gemini.json create mode 100644 tests/llm/golden/snapshots/groq.json create mode 100644 tests/llm/golden/snapshots/openrouter_claude.json create mode 100644 tests/llm/golden/snapshots/openrouter_non_claude.json create mode 100644 tests/llm/golden/snapshots/remote.json create mode 100644 tests/llm/golden/snapshots/unruled_anthropic.json create mode 100644 tests/llm/golden/snapshots/unruled_bedrock.json create mode 100644 tests/llm/golden/snapshots/unruled_gemini.json create mode 100644 tests/llm/golden/snapshots/unruled_grok.json create mode 100644 tests/llm/golden/snapshots/unruled_openai.json create mode 100644 tests/llm/golden/snapshots/unruled_openrouter_claude.json create mode 100644 tests/llm/test_reasoning_rules.py create mode 100644 tests/llm/test_reliability.py diff --git a/agent_core/core/impl/llm/cache/gemini.py b/agent_core/core/impl/llm/cache/gemini.py index fc06a813..1b1376a9 100644 --- a/agent_core/core/impl/llm/cache/gemini.py +++ b/agent_core/core/impl/llm/cache/gemini.py @@ -10,7 +10,7 @@ import hashlib import logging import time -from typing import Any, Dict, TYPE_CHECKING +from typing import Any, Dict, Optional, TYPE_CHECKING from .config import get_cache_config @@ -80,6 +80,8 @@ def get_or_create_cache( call_type: str, temperature: float, max_tokens: int, + thinking_budget: Optional[int] = None, + thinking_level: Optional[str] = None, ) -> Dict[str, Any]: """Get response using explicit cache, creating cache if needed. @@ -89,10 +91,19 @@ def get_or_create_cache( call_type: Type of LLM call (e.g., "reasoning", "action_selection"). temperature: Sampling temperature. max_tokens: Maximum output tokens. + thinking_budget: Reasoning token budget (Gemini 2.5), forwarded + to every generation call. + thinking_level: Reasoning level (Gemini 3.x), forwarded to every + generation call. Returns: Response dict with tokens_used, content, cached_tokens, etc. """ + thinking: Dict[str, Any] = { + "thinking_budget": thinking_budget, + "thinking_level": thinking_level, + } + # Check if system prompt is large enough for explicit caching # Gemini requires at least 1024 tokens; skip explicit cache if too small estimated_tokens = self._estimate_tokens(system_prompt) @@ -109,6 +120,7 @@ def get_or_create_cache( temperature=temperature, max_output_tokens=max_tokens, json_mode=True, + **thinking, ) cache_key = self._make_cache_key(system_prompt, call_type) @@ -132,6 +144,7 @@ def get_or_create_cache( temperature=temperature, max_output_tokens=max_tokens, json_mode=True, + **thinking, ) except Exception as e: logger.warning( @@ -166,6 +179,7 @@ def get_or_create_cache( temperature=temperature, max_output_tokens=max_tokens, json_mode=True, + **thinking, ) except Exception as e: logger.warning( @@ -185,6 +199,7 @@ def get_or_create_cache( temperature=temperature, max_output_tokens=max_tokens, json_mode=True, + **thinking, ) def invalidate_cache(self, system_prompt: str, call_type: str) -> None: diff --git a/agent_core/core/impl/llm/interface.py b/agent_core/core/impl/llm/interface.py index c5965b9d..7a97fd51 100644 --- a/agent_core/core/impl/llm/interface.py +++ b/agent_core/core/impl/llm/interface.py @@ -42,6 +42,7 @@ RecordLLMCallHook, ) from agent_core.core.impl.llm import transports as _transports +from agent_core.core.models.reasoning import ReasoningDecision, resolve_reasoning from agent_core.core.models.registry import ( get_registry as _get_registry, session_cc_providers as _session_cc_providers, @@ -227,6 +228,9 @@ def __init__( # multi-provider outage terminates instead of nesting # primary -> fb -> fb-of-fb recursion. self._is_fallback_instance = False + # (provider, model, auth_mode) whose reasoning default was last + # logged, so the decision is logged once per model, not per call. + self._reasoning_logged_for: Optional[tuple] = None # Defer imports to avoid circular dependency from app.models.factory import ModelFactory @@ -776,7 +780,8 @@ def _check_context_fits( from app.config import get_context_window window = get_context_window() - budget = window - self.max_tokens + reserved = self._output_reservation() + budget = window - reserved total = count_tokens(system_prompt or "") if messages: @@ -795,10 +800,43 @@ def _check_context_fits( if total > budget: raise LLMContextOverflowError( f"Request of ~{total} input tokens exceeds the {budget}-token budget " - f"({window} window - {self.max_tokens} reserved for output) for " + f"({window} window - {reserved} reserved for output) for " f"{self.provider}/{self.model}." ) + def reasoning_decision(self) -> Optional[ReasoningDecision]: + """Reasoning default for the current provider and model. + + None means the model has no rule in agent_core/core/models/reasoning.py + and its requests carry no reasoning parameter. Resolved on every call + so a model switch (reinitialize) or a fallback interface uses its own + row; logged once per distinct model. + """ + decision = resolve_reasoning(self.provider, self.model, self._auth_mode) + log_key = (self.provider, self.model, self._auth_mode) + if log_key != self._reasoning_logged_for: + self._reasoning_logged_for = log_key + if decision is None: + logger.info( + f"[REASONING] {self.provider}/{self.model}: no reasoning rule " + f"for this model, reasoning parameters not sent" + ) + else: + logger.info(f"[REASONING] {decision.key}: {decision.describe()}") + return decision + + def _output_reservation(self) -> int: + """Output tokens a request reserves out of the context window. + + Requests with a reasoning default carry a larger output cap (reasoning + tokens count against it), so the window check reserves that cap; + otherwise the provider would reject the request for size instead. + """ + decision = self.reasoning_decision() + if decision is None: + return self.max_tokens + return decision.output_cap(self.max_tokens) + def _generate_response_sync( self, system_prompt: Optional[str] = None, diff --git a/agent_core/core/impl/llm/reasoning_wire.py b/agent_core/core/impl/llm/reasoning_wire.py new file mode 100644 index 00000000..9203801b --- /dev/null +++ b/agent_core/core/impl/llm/reasoning_wire.py @@ -0,0 +1,87 @@ +# -*- coding: utf-8 -*- +"""Render a reasoning default into request fields, one function per transport. + +The table and the level policy live in agent_core/core/models/reasoning.py; +this module only knows how each transport spells a ``ReasoningDecision``. +Callers skip these functions entirely when a model has no rule, so a model +without a rule never gains a field here. + +A decision whose wire the transport cannot express is a bug in the rules +table (a row filed under the wrong provider), so it raises instead of being +dropped silently. tests/llm/test_reasoning_rules.py checks every row against +``TRANSPORT_WIRES`` so this never reaches production. +""" + +from __future__ import annotations + +from typing import Any, Dict, FrozenSet, Mapping, Tuple + +from agent_core.core.models.reasoning import ReasoningDecision, ReasoningWire + +#: Reasoning wires each transport (ProviderProfile.wire) can express. +TRANSPORT_WIRES: Mapping[str, FrozenSet[ReasoningWire]] = { + "chat_completions": frozenset( + { + ReasoningWire.EFFORT, + ReasoningWire.OPENROUTER_EFFORT, + ReasoningWire.OPENROUTER_BUDGET, + } + ), + "anthropic_messages": frozenset( + {ReasoningWire.ANTHROPIC_ADAPTIVE, ReasoningWire.ANTHROPIC_BUDGET} + ), + "bedrock_converse": frozenset( + {ReasoningWire.ANTHROPIC_ADAPTIVE, ReasoningWire.ANTHROPIC_BUDGET} + ), + "gemini_native": frozenset( + {ReasoningWire.GEMINI_LEVEL, ReasoningWire.GEMINI_BUDGET} + ), +} + + +def _unsupported(decision: ReasoningDecision, transport: str) -> ValueError: + return ValueError( + f"Reasoning rule {decision.key!r} uses wire {decision.wire.value!r}, " + f"which the {transport} transport cannot send. The row is filed under " + f"the wrong provider in agent_core/core/models/reasoning.py." + ) + + +def chat_completions_fields( + decision: ReasoningDecision, +) -> Tuple[Dict[str, Any], Dict[str, Any]]: + """Fields for a Chat Completions request: (top-level kwargs, extra_body).""" + if decision.wire is ReasoningWire.EFFORT: + return {"reasoning_effort": decision.level}, {} + if decision.wire is ReasoningWire.OPENROUTER_EFFORT: + return {}, {"reasoning": {"effort": decision.level}} + if decision.wire is ReasoningWire.OPENROUTER_BUDGET: + return {}, {"reasoning": {"max_tokens": decision.budget_tokens}} + raise _unsupported(decision, "chat_completions") + + +def anthropic_fields(decision: ReasoningDecision) -> Dict[str, Any]: + """Anthropic Messages fields. + + The same dict is the ``additionalModelRequestFields`` of a Bedrock + Converse request for Claude, which forwards these keys to the model. + """ + if decision.wire is ReasoningWire.ANTHROPIC_ADAPTIVE: + return { + "thinking": {"type": "adaptive"}, + "output_config": {"effort": decision.level}, + } + if decision.wire is ReasoningWire.ANTHROPIC_BUDGET: + return { + "thinking": {"type": "enabled", "budget_tokens": decision.budget_tokens} + } + raise _unsupported(decision, "anthropic_messages/bedrock_converse") + + +def gemini_thinking_kwargs(decision: ReasoningDecision) -> Dict[str, Any]: + """Keyword arguments for the GeminiClient text-generation methods.""" + if decision.wire is ReasoningWire.GEMINI_LEVEL: + return {"thinking_level": decision.level} + if decision.wire is ReasoningWire.GEMINI_BUDGET: + return {"thinking_budget": decision.budget_tokens} + raise _unsupported(decision, "gemini_native") diff --git a/agent_core/core/impl/llm/transports/anthropic_messages.py b/agent_core/core/impl/llm/transports/anthropic_messages.py index dd031ead..56639b0f 100644 --- a/agent_core/core/impl/llm/transports/anthropic_messages.py +++ b/agent_core/core/impl/llm/transports/anthropic_messages.py @@ -10,10 +10,14 @@ from typing import Any, Dict, List, Optional from agent_core.decorators import profile, OperationCategory +from agent_core.core.impl.llm import reasoning_wire from agent_core.core.impl.llm.cache import get_cache_config, get_cache_metrics from agent_core.core.impl.llm.errors import classify_llm_error from agent_core.utils.logger import logger +# Anthropic requires max_tokens; 16384 (Claude 4 default) avoids truncation. +_DEFAULT_MAX_TOKENS = 16384 + @profile("llm_anthropic_call", OperationCategory.LLM) def generate( @@ -71,11 +75,20 @@ def generate( if not iface._anthropic_client: raise RuntimeError("Anthropic client was not initialised.") - # Build the message - use pre-built messages for multi-turn, or single-turn - # Anthropic requires max_tokens; use 16384 (Claude 4 default) to avoid truncation + # Reasoning default for this exact model (None: no rule, so the + # request is shaped exactly as it was before reasoning defaults). + reasoning = iface.reasoning_decision() + + # Build the message - use pre-built messages for multi-turn, or single-turn. + # Thinking tokens count against max_tokens, so a reasoning default + # raises it (staying below the SDK's non-streaming ceiling). message_kwargs: Dict[str, Any] = { "model": iface.model, - "max_tokens": 16384, + "max_tokens": ( + _DEFAULT_MAX_TOKENS + if reasoning is None + else reasoning.output_cap(_DEFAULT_MAX_TOKENS) + ), "messages": messages if messages is not None else [ @@ -109,7 +122,14 @@ def generate( # Short prompt - use simple string format (no caching) message_kwargs["system"] = system_prompt - message_kwargs["extra_body"] = {"temperature": iface.temperature} + if reasoning is not None: + message_kwargs.update(reasoning_wire.anthropic_fields(reasoning)) + + # Thinking is incompatible with temperature on Claude 4.5/4.6, and + # Claude 4.7+ rejects any non-default temperature, so rows with + # reasoning drop it (the model then uses its default). + if reasoning is None or not reasoning.omit_temperature: + message_kwargs["extra_body"] = {"temperature": iface.temperature} response = iface._anthropic_client.messages.create(**message_kwargs) diff --git a/agent_core/core/impl/llm/transports/bedrock_converse.py b/agent_core/core/impl/llm/transports/bedrock_converse.py index 202231c8..bbf3abe2 100644 --- a/agent_core/core/impl/llm/transports/bedrock_converse.py +++ b/agent_core/core/impl/llm/transports/bedrock_converse.py @@ -11,6 +11,7 @@ from typing import Any, Dict, List, Optional from agent_core.decorators import profile, OperationCategory +from agent_core.core.impl.llm import reasoning_wire from agent_core.core.impl.llm.cache import get_cache_config, get_cache_metrics from agent_core.core.impl.llm.errors import classify_llm_error from agent_core.utils.logger import logger @@ -70,6 +71,10 @@ def generate( else [{"role": "user", "content": [{"text": user_prompt}]}] ) + # Reasoning default for this exact model (None: no rule, so the + # request is shaped exactly as it was before reasoning defaults). + reasoning = iface.reasoning_decision() + converse_kwargs: Dict[str, Any] = { "modelId": iface.model, "messages": converse_messages, @@ -78,6 +83,19 @@ def generate( "maxTokens": iface.max_tokens, }, } + if reasoning is not None: + # Claude on Converse takes the Messages-API thinking/effort keys + # through additionalModelRequestFields. Thinking tokens count + # against maxTokens, and thinking is incompatible with (4.5/4.6) + # or rejects (4.7+) a non-default temperature. + converse_kwargs["additionalModelRequestFields"] = ( + reasoning_wire.anthropic_fields(reasoning) + ) + converse_kwargs["inferenceConfig"]["maxTokens"] = reasoning.output_cap( + iface.max_tokens + ) + if reasoning.omit_temperature: + del converse_kwargs["inferenceConfig"]["temperature"] if system_prompt: # When messages already carry a cachePoint (multi-turn first diff --git a/agent_core/core/impl/llm/transports/chat_completions.py b/agent_core/core/impl/llm/transports/chat_completions.py index 1420b0ce..f0871153 100644 --- a/agent_core/core/impl/llm/transports/chat_completions.py +++ b/agent_core/core/impl/llm/transports/chat_completions.py @@ -19,6 +19,7 @@ import requests from agent_core.decorators import profile, OperationCategory +from agent_core.core.impl.llm import reasoning_wire from agent_core.core.impl.llm.cache import get_cache_config, get_cache_metrics from agent_core.core.impl.llm.errors import classify_llm_error, provider_display_name from agent_core.core.models.registry import ( @@ -114,9 +115,14 @@ def generate_openai( "model": iface.model, "messages": messages, } + # Reasoning default for this exact model (None: no rule, so the + # request is shaped exactly as it was before reasoning defaults). + reasoning = iface.reasoning_decision() _profile = _get_registry().get(iface.provider) _temp = _resolve_temperature(_profile, iface.temperature) - if _temp is not _OMIT_TEMPERATURE: + if _temp is not _OMIT_TEMPERATURE and not ( + reasoning is not None and reasoning.omit_temperature + ): request_kwargs["temperature"] = _temp # Output tokens: cap the VALUE to the provider's output limit (several @@ -124,7 +130,13 @@ def generate_openai( # when it's exceeded), and pick the FIELD NAME per provider policy # (profile.uses_max_completion_tokens: OpenAI/Cerebras/MiniMax/Groq # take 'max_completion_tokens'; everyone else legacy 'max_tokens'). - _max_tokens_value = iface.max_tokens + # Reasoning tokens count against this cap, so a reasoning default + # raises it (never lowers it) to leave room for the answer. + _max_tokens_value = ( + iface.max_tokens + if reasoning is None + else reasoning.output_cap(iface.max_tokens) + ) if _profile is not None and _profile.max_output_tokens: _max_tokens_value = min(_max_tokens_value, _profile.max_output_tokens) uses_max_completion_tokens = ( @@ -199,6 +211,13 @@ def generate_openai( f"[OPENROUTER] Anthropic cache_control: {cache_control} (model={iface.model})" ) + if reasoning is not None: + reasoning_top, reasoning_extra = reasoning_wire.chat_completions_fields( + reasoning + ) + request_kwargs.update(reasoning_top) + extra_body.update(reasoning_extra) + if extra_body: request_kwargs["extra_body"] = extra_body diff --git a/agent_core/core/impl/llm/transports/gemini_native.py b/agent_core/core/impl/llm/transports/gemini_native.py index 0254f738..4ddc8305 100644 --- a/agent_core/core/impl/llm/transports/gemini_native.py +++ b/agent_core/core/impl/llm/transports/gemini_native.py @@ -10,6 +10,7 @@ from typing import Any, Dict, List, Optional from agent_core.decorators import profile, OperationCategory +from agent_core.core.impl.llm import reasoning_wire from agent_core.core.impl.llm.cache import get_cache_config, get_cache_metrics from agent_core.core.impl.llm.errors import classify_llm_error from agent_core.utils.logger import logger @@ -52,11 +53,24 @@ def generate( # Per-call reasoning cap, set by callers that pass thinking_budget (e.g. the # entity-judge pipeline). Rides the shared per-call context so no transport - # signature changes; None for every ordinary call, in which case Gemini's - # default thinking behaviour is unchanged. + # signature changes. An explicit per-call budget wins over the model's + # reasoning default; otherwise the default for this exact model applies + # (None: no rule, so Gemini's own default thinking is left untouched). from agent_core.core.impl.llm.interface import _llm_call_ctx - thinking_budget = (_llm_call_ctx.get() or {}).get("thinking_budget") + caller_budget = (_llm_call_ctx.get() or {}).get("thinking_budget") + reasoning = iface.reasoning_decision() if caller_budget is None else None + if caller_budget is not None: + thinking: Dict[str, Any] = {"thinking_budget": caller_budget} + elif reasoning is not None: + thinking = reasoning_wire.gemini_thinking_kwargs(reasoning) + else: + thinking = {} + # Thinking tokens count against maxOutputTokens, so a reasoning default + # raises the cap (never lowers it) to leave room for the answer. + max_output_tokens = ( + iface.max_tokens if reasoning is None else reasoning.output_cap(iface.max_tokens) + ) token_count_input = token_count_output = 0 cached_tokens = 0 @@ -85,8 +99,9 @@ def generate( contents=contents_override, system_prompt=system_prompt, temperature=iface.temperature, - max_output_tokens=iface.max_tokens, + max_output_tokens=max_output_tokens, json_mode=json_mode, + **thinking, ) else: # Use explicit caching when: @@ -116,7 +131,8 @@ def generate( user_prompt=user_prompt, call_type=call_type, temperature=iface.temperature, - max_tokens=iface.max_tokens, + max_tokens=max_output_tokens, + **thinking, ) else: # Fall back to implicit caching (or no caching for short prompts) @@ -125,9 +141,9 @@ def generate( prompt=user_prompt, system_prompt=system_prompt, temperature=iface.temperature, - max_output_tokens=iface.max_tokens, + max_output_tokens=max_output_tokens, json_mode=json_mode, - thinking_budget=thinking_budget, + **thinking, ) # Extract response data diff --git a/agent_core/core/llm/google_gemini_client.py b/agent_core/core/llm/google_gemini_client.py index db9f6290..98e2593b 100644 --- a/agent_core/core/llm/google_gemini_client.py +++ b/agent_core/core/llm/google_gemini_client.py @@ -38,6 +38,25 @@ def _normalise_model_name(model: str) -> str: return model if model.startswith("models/") else f"models/{model}" +def _thinking_config( + thinking_budget: Optional[int], thinking_level: Optional[str] +) -> Optional[Dict[str, Any]]: + """Build ``generationConfig.thinkingConfig``, or None to omit it. + + ``thinkingBudget`` is the Gemini 2.5 knob and ``thinkingLevel`` the + Gemini 3.x one; the API rejects a request that sets both with a 400. + """ + if thinking_budget is not None and thinking_level is not None: + raise ValueError( + "Gemini accepts either thinking_budget or thinking_level, not both." + ) + if thinking_budget is not None: + return {"thinkingBudget": thinking_budget} + if thinking_level is not None: + return {"thinkingLevel": thinking_level} + return None + + class GeminiClient: """Lightweight REST client for Gemini models. @@ -94,6 +113,7 @@ def generate_text( max_output_tokens: Optional[int] = None, json_mode: bool = False, thinking_budget: Optional[int] = None, + thinking_level: Optional[str] = None, ) -> Dict[str, Any]: """Generate text for a purely textual prompt. @@ -121,6 +141,9 @@ def generate_text( (and can exhaust maxOutputTokens on thoughts alone, emitting no text — finishReason=MAX_TOKENS, parts_count=0). Set it to reserve output room for the actual answer. + thinking_level: Optional reasoning level for Gemini 3.x thinking + models (``thinkingLevel``). Mutually exclusive with + ``thinking_budget``. Returns: Dict with generation results and token counts @@ -139,8 +162,9 @@ def generate_text( generation_config["maxOutputTokens"] = max_output_tokens if json_mode: generation_config["responseMimeType"] = "application/json" - if thinking_budget is not None: - generation_config["thinkingConfig"] = {"thinkingBudget": thinking_budget} + thinking_config = _thinking_config(thinking_budget, thinking_level) + if thinking_config is not None: + generation_config["thinkingConfig"] = thinking_config payload: Dict[str, Any] = {"contents": contents} if system_prompt: @@ -181,6 +205,8 @@ def generate_text_multiturn( temperature: Optional[float] = None, max_output_tokens: Optional[int] = None, json_mode: bool = False, + thinking_budget: Optional[int] = None, + thinking_level: Optional[str] = None, ) -> Dict[str, Any]: """Generate text from a pre-built multi-turn `contents` array. @@ -199,6 +225,10 @@ def generate_text_multiturn( temperature: Sampling temperature. max_output_tokens: Output token cap. json_mode: Force JSON response. + thinking_budget: Reasoning token budget (Gemini 2.5), see + ``generate_text``. + thinking_level: Reasoning level (Gemini 3.x), see + ``generate_text``. Returns: Same shape as ``generate_text``. @@ -210,6 +240,9 @@ def generate_text_multiturn( generation_config["maxOutputTokens"] = max_output_tokens if json_mode: generation_config["responseMimeType"] = "application/json" + thinking_config = _thinking_config(thinking_budget, thinking_level) + if thinking_config is not None: + generation_config["thinkingConfig"] = thinking_config payload: Dict[str, Any] = {"contents": contents} if system_prompt: @@ -425,6 +458,8 @@ def generate_text_with_cache( temperature: Optional[float] = None, max_output_tokens: Optional[int] = None, json_mode: bool = False, + thinking_budget: Optional[int] = None, + thinking_level: Optional[str] = None, ) -> Dict[str, Any]: """Generate text using an explicit cache. @@ -438,6 +473,10 @@ def generate_text_with_cache( temperature: Sampling temperature max_output_tokens: Maximum output tokens json_mode: If True, enforce JSON output format + thinking_budget: Reasoning token budget (Gemini 2.5), see + ``generate_text``. + thinking_level: Reasoning level (Gemini 3.x), see + ``generate_text``. Returns: Dict with tokens_used, content, prompt_tokens, completion_tokens, cached_tokens @@ -456,6 +495,9 @@ def generate_text_with_cache( generation_config["maxOutputTokens"] = max_output_tokens if json_mode: generation_config["responseMimeType"] = "application/json" + thinking_config = _thinking_config(thinking_budget, thinking_level) + if thinking_config is not None: + generation_config["thinkingConfig"] = thinking_config payload: Dict[str, Any] = { "contents": contents, diff --git a/agent_core/core/models/chatgpt_subscription_client.py b/agent_core/core/models/chatgpt_subscription_client.py index d29c9493..c2e82fef 100644 --- a/agent_core/core/models/chatgpt_subscription_client.py +++ b/agent_core/core/models/chatgpt_subscription_client.py @@ -15,8 +15,10 @@ - ``store: false`` ("Store must be set to false") - ``stream: true`` ("Stream must be set to true"); aggregated below -- ``reasoning.effort: `` ("none" for non-codex 5.1/5.2, - "low" for codex variants — "minimal" is rejected by the backend) +- ``reasoning.effort``: the caller's ``reasoning_effort`` (resolved per model + by agent_core/core/models/reasoning.py under the ``openai_subscription`` + surface), or ``"medium"`` for a model without a rule. The block itself is + required, so it is never omitted. - ``reasoning.summary: "auto"`` - ``include: ["reasoning.encrypted_content"]`` (mandatory under ``store=false`` so the model can keep its own @@ -161,14 +163,23 @@ def __init__( # silently honoring "best-effort" semantics is fine for fields that just # don't apply at this backend (e.g. ``max_tokens`` becomes "let the # server decide" rather than a hard failure). -def _codex_reasoning_config(_model: str) -> Dict[str, str]: +#: Effort for a model the reasoning table has no row for. The Codex backend +#: requires a ``reasoning`` block on every request, so unlike every other +#: provider the parameter cannot simply be left out; "medium" is the Codex +#: CLI default and what this translator always sent before per-model rules. +_CODEX_UNRULED_EFFORT = "medium" + + +def _codex_reasoning_config(requested_effort: Any) -> Dict[str, str]: """Build the ``reasoning`` block Codex requires on every request. - "medium" effort matches the Codex CLI default — fast enough for - JSON action-decision loops, deliberate enough that the model - follows instruction-following. ``"auto"`` summary matches the CLI. + ``requested_effort`` is the caller's Chat-Completions + ``reasoning_effort``: the per-model default the chat_completions + transport resolved, or None when the model has no rule. + ``"auto"`` summary matches the Codex CLI. """ - return {"effort": "medium", "summary": "auto"} + effort = requested_effort if requested_effort else _CODEX_UNRULED_EFFORT + return {"effort": str(effort), "summary": "auto"} def _extract_instructions( @@ -231,7 +242,7 @@ def _translate_request( out["input"] = _normalize_messages(rest) # ``reasoning`` is REQUIRED for every gpt-5.x model on Codex. - out["reasoning"] = _codex_reasoning_config(model) + out["reasoning"] = _codex_reasoning_config(kwargs.get("reasoning_effort")) # ``text.verbosity`` is required by the Codex backend to know how # long a response to produce. The reference impl always sets it; diff --git a/agent_core/core/models/factory.py b/agent_core/core/models/factory.py index 2006e5d8..e360c96b 100644 --- a/agent_core/core/models/factory.py +++ b/agent_core/core/models/factory.py @@ -11,8 +11,10 @@ try: import boto3 # type: ignore[import] + from botocore.config import Config as _BotocoreConfig # type: ignore[import] except ImportError: # pragma: no cover — boto3 is an optional extra boto3 = None # type: ignore[assignment] + _BotocoreConfig = None # type: ignore[assignment] from agent_core.core.models.types import InterfaceType from agent_core.core.models.provider_config import PROVIDER_CONFIG @@ -24,6 +26,12 @@ logger = logging.getLogger(__name__) +# Read timeout for Bedrock Converse calls. botocore's default is 60 seconds, +# which a Claude response with thinking can exceed; AWS documents a +# 60-minute server-side limit for Claude 4 and later. 600 seconds matches the +# Anthropic SDK's default request timeout used by the direct transport. +_BEDROCK_READ_TIMEOUT_SECONDS = 600 + # Derived from provider profiles (Phase 1, docs/PROVIDER_LAYER_CATCHUP.md). # OpenRouter proxy routing exists because some direct APIs are geo-restricted # for most international users; the per-provider data lives on the profiles. @@ -618,7 +626,12 @@ def create( ) try: - client_kwargs = {"region_name": region} + client_kwargs = { + "region_name": region, + "config": _BotocoreConfig( + read_timeout=_BEDROCK_READ_TIMEOUT_SECONDS + ), + } if access_key and secret_key: client_kwargs["aws_access_key_id"] = access_key client_kwargs["aws_secret_access_key"] = secret_key diff --git a/agent_core/core/models/reasoning.py b/agent_core/core/models/reasoning.py new file mode 100644 index 00000000..3f2301ce --- /dev/null +++ b/agent_core/core/models/reasoning.py @@ -0,0 +1,814 @@ +# -*- coding: utf-8 -*- +"""Default reasoning effort ("thinking") per model. + +Why this exists +--------------- +Reasoning models need an explicit request parameter to think at a useful +depth, and several of them do no reasoning at all when it is omitted +(OpenAI gpt-5.1 through gpt-5.4 default to ``none``; Claude Sonnet 4.6 and +Opus 4.6-4.8 run without thinking unless it is requested). The parameter is +also model-specific: its name, shape, and accepted values differ between +models of the SAME provider, and a value a model does not accept is a hard +HTTP 400 on the first request. + +So defaults live in a hard-coded table keyed by provider and EXACT model id +(no prefix or substring matching). A model that is not in the table gets no +reasoning parameter at all, and its request stays byte-identical to the one +sent before this module existed (pinned by the ``unruled_*`` golden payload +snapshots under tests/llm/golden/). + +Choosing the level +------------------ +``ReasoningRule.levels`` lists the levels that turn reasoning on, weakest +first. "Off" values (``none``, ``disabled``, ``minimal``) and values a +provider silently maps onto another level are left out, so the tuple holds +only levels that behave differently. The default level is: + +1. one level below the model's strongest level (``LEVELS_BELOW_MAX``); +2. never stronger than ``LEVEL_CEILING`` (``high``); +3. but if that is weaker than the level the provider already applies when + the parameter is omitted (``provider_default``), the model's strongest + level is used instead, so a default is never lowered. + +Token-budget models (Claude 4.5, Gemini 2.5) have no named levels; their +rows name four budget rungs ``BUDGET_RUNGS`` so the same rule picks the +third rung ("high"). + +Adding a model +-------------- +Add its exact id (every alias and dated snapshot the provider accepts) to a +rule below, with the ordered levels from the provider's documentation and +the documentation URL in ``source``. The unit tests in +tests/llm/test_reasoning_rules.py validate every row. + +The keys follow the ``/`` shape of the model catalog +planned in docs/PROVIDER_LAYER_CATCHUP.md section 10, so the rows can move +into that catalog as a column without a rewrite. +""" + +from __future__ import annotations + +from dataclasses import dataclass, replace +from enum import Enum +from typing import Dict, Mapping, Optional, Tuple + +# ─────────────────────────────── policy ─────────────────────────────── + +#: How many levels below the model's strongest level the default sits. +LEVELS_BELOW_MAX = 1 + +#: The strongest level ever chosen as a default (unless the provider's own +#: default is already stronger; see rule 3 in the module docstring). +LEVEL_CEILING = "high" + +#: Every level name any provider uses, weakest first. Rows must use these. +KNOWN_LEVELS: Tuple[str, ...] = ("minimal", "low", "medium", "high", "xhigh", "max") + +#: Rung names for token-budget models, weakest first. +BUDGET_RUNGS: Tuple[str, ...] = ("low", "medium", "high", "max") + +#: The auth mode whose OpenAI requests go to the ChatGPT-subscription Codex +#: backend, which serves a different model catalogue with different levels. +SUBSCRIPTION_AUTH_MODE = "subscription" + + +class ReasoningWire(str, Enum): + """How a reasoning default is expressed in a request.""" + + #: Top-level ``reasoning_effort: `` on a Chat Completions request. + #: The ChatGPT-subscription translator maps it to ``reasoning.effort``. + EFFORT = "effort" + #: Anthropic ``thinking: {"type": "adaptive"}`` plus + #: ``output_config: {"effort": }`` (Claude 4.6 and later). + ANTHROPIC_ADAPTIVE = "anthropic_adaptive" + #: Anthropic ``thinking: {"type": "enabled", "budget_tokens": }`` + #: (Claude 4.5 generation). + ANTHROPIC_BUDGET = "anthropic_budget" + #: Gemini ``thinkingConfig: {"thinkingLevel": }`` (Gemini 3.x). + GEMINI_LEVEL = "gemini_level" + #: Gemini ``thinkingConfig: {"thinkingBudget": }`` (Gemini 2.5). + GEMINI_BUDGET = "gemini_budget" + #: OpenRouter ``reasoning: {"effort": }``. + OPENROUTER_EFFORT = "openrouter_effort" + #: OpenRouter ``reasoning: {"max_tokens": }``. + OPENROUTER_BUDGET = "openrouter_budget" + + +#: Wires whose rows carry token budgets instead of named levels. +BUDGET_WIRES = frozenset( + { + ReasoningWire.ANTHROPIC_BUDGET, + ReasoningWire.GEMINI_BUDGET, + ReasoningWire.OPENROUTER_BUDGET, + } +) + + +@dataclass(frozen=True) +class ReasoningRule: + """Reasoning capabilities of one model (or of models that share them).""" + + #: How the default is expressed in the request. + wire: ReasoningWire + #: Levels that turn reasoning on, weakest first (``BUDGET_RUNGS`` for + #: budget wires). + levels: Tuple[str, ...] + #: Level the provider applies when the parameter is omitted, when that + #: is one of ``levels``. None when omitting it means no reasoning, a + #: provider-chosen dynamic amount, or an "off" value. + provider_default: Optional[str] + #: Documentation the row was taken from. + source: str + #: Budget-wire rows: token budget per rung, aligned with ``levels``. + budgets: Tuple[int, ...] = () + #: Output-token cap to request while reasoning is on. Reasoning tokens + #: count against the provider's output cap, so it must leave room for + #: the answer. 0 keeps the caller's cap unchanged. + output_tokens: int = 0 + #: The model rejects ``temperature`` while reasoning is on. + omit_temperature: bool = False + + +@dataclass(frozen=True) +class ReasoningDecision: + """The reasoning default resolved for one (provider, model) pair.""" + + #: ``/`` of the row that produced this decision. + key: str + wire: ReasoningWire + level: str + #: Token budget for budget wires, else None. + budget_tokens: Optional[int] + output_tokens: int + omit_temperature: bool + + def output_cap(self, base: int) -> int: + """Output-token cap for a request whose cap is otherwise ``base``. + + Never lowers ``base``: reasoning only ever adds room. + """ + return max(base, self.output_tokens) + + def describe(self) -> str: + """One-line human-readable summary for logs.""" + if self.budget_tokens is not None: + return f"{self.wire.value} budget={self.budget_tokens} ({self.level})" + return f"{self.wire.value} level={self.level}" + + +def target_level(rule: ReasoningRule) -> str: + """Pick the default level for ``rule`` (see the module docstring).""" + levels = rule.levels + index = max(len(levels) - 1 - LEVELS_BELOW_MAX, 0) + if LEVEL_CEILING in levels: + index = min(index, levels.index(LEVEL_CEILING)) + if rule.provider_default in levels and index < levels.index(rule.provider_default): + index = len(levels) - 1 + return levels[index] + + +def reasoning_surface(provider: str, auth_mode: str) -> str: + """Table surface for a provider and auth mode. + + OpenAI requests made with a ChatGPT-subscription login go to the Codex + backend, whose model catalogue and levels differ from the public API. + """ + if provider == "openai" and auth_mode == SUBSCRIPTION_AUTH_MODE: + return "openai_subscription" + return provider + + +def resolve_reasoning( + provider: Optional[str], + model: Optional[str], + auth_mode: str = "api_key", +) -> Optional[ReasoningDecision]: + """Resolve the reasoning default for a model, or None if it has no row.""" + if not provider or not model: + return None + surface = reasoning_surface(provider, auth_mode) + rule = REASONING_RULES.get(surface, {}).get(model) + if rule is None: + return None + level = target_level(rule) + budget = ( + rule.budgets[rule.levels.index(level)] if rule.wire in BUDGET_WIRES else None + ) + return ReasoningDecision( + key=f"{surface}/{model}", + wire=rule.wire, + level=level, + budget_tokens=budget, + output_tokens=rule.output_tokens, + omit_temperature=rule.omit_temperature, + ) + + +def _models(rule: ReasoningRule, *model_ids: str) -> Dict[str, ReasoningRule]: + return {model_id: rule for model_id in model_ids} + + +def _surface(*groups: Dict[str, ReasoningRule]) -> Dict[str, ReasoningRule]: + """Merge one provider's ``_models`` groups, refusing duplicate ids. + + A plain dict merge would let a second listing of an id silently replace + the first, sending a value its real model may reject. + """ + merged: Dict[str, ReasoningRule] = {} + for group in groups: + for model_id, rule in group.items(): + if model_id in merged: + raise ValueError( + f"Model id {model_id!r} is listed twice in the reasoning table." + ) + merged[model_id] = rule + return merged + + +def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: + """A Bedrock base model id plus its documented inference-profile ids. + + ``profile_prefixes`` are the geo/global prefixes the model's AWS model + card lists (for example ``us`` gives ``us.``); each resulting + id is matched exactly like any other row. + """ + return (base_id, *(f"{prefix}.{base_id}" for prefix in profile_prefixes)) + + +# ─────────────────────────────── rules ──────────────────────────────── +# +# Output caps. Non-streaming Anthropic requests are refused by the Anthropic +# SDK above 21,333 max_tokens (anthropic/_base_client.py +# _calculate_nonstreaming_timeout), so Anthropic-family rows stay just below +# it. OpenAI-style effort rows use 32,000: OpenAI recommends reserving at +# least 25,000 tokens for reasoning plus output +# (https://developers.openai.com/api/docs/guides/reasoning). + +_ANTHROPIC_OUTPUT_TOKENS = 21_000 +_EFFORT_OUTPUT_TOKENS = 32_000 + +# ── OpenAI public API (Chat Completions ``reasoning_effort``) ── +# Values and defaults: https://developers.openai.com/api/docs/models/. +# Models before gpt-5.1 reject "none"; gpt-5.1+ reject "minimal"; xhigh +# arrived with gpt-5.2 and max with gpt-5.6. Temperature is rejected whenever +# effort is not "none" (the OpenAI profile already omits it). + +_OPENAI_DOCS = "https://developers.openai.com/api/docs/models" + +_OPENAI_GPT5 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high"), + provider_default="medium", + source=f"{_OPENAI_DOCS}/gpt-5", + output_tokens=_EFFORT_OUTPUT_TOKENS, + omit_temperature=True, +) +_OPENAI_O_SERIES = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high"), + provider_default="medium", + source="https://developers.openai.com/api/docs/guides/reasoning", + output_tokens=_EFFORT_OUTPUT_TOKENS, + omit_temperature=True, +) +_OPENAI_GPT51 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high"), + provider_default=None, # "none": no reasoning + source=f"{_OPENAI_DOCS}/gpt-5.1", + output_tokens=_EFFORT_OUTPUT_TOKENS, + omit_temperature=True, +) +_OPENAI_GPT52_TO_54 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high", "xhigh"), + provider_default=None, # "none": no reasoning + source=f"{_OPENAI_DOCS}/gpt-5.2", + output_tokens=_EFFORT_OUTPUT_TOKENS, + omit_temperature=True, +) +_OPENAI_GPT55 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high", "xhigh"), + provider_default="medium", + source=f"{_OPENAI_DOCS}/gpt-5.5", + output_tokens=_EFFORT_OUTPUT_TOKENS, + omit_temperature=True, +) +_OPENAI_GPT56_PLUS = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high", "xhigh", "max"), + provider_default="medium", + source=f"{_OPENAI_DOCS}/gpt-6-sol", + output_tokens=_EFFORT_OUTPUT_TOKENS, + omit_temperature=True, +) +#: GPT-6 Astra rejects "none" and documents no default. +_OPENAI_GPT6_ASTRA = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high", "xhigh", "max"), + provider_default=None, + source=f"{_OPENAI_DOCS}/gpt-6-astra", + output_tokens=_EFFORT_OUTPUT_TOKENS, + omit_temperature=True, +) + +# ── ChatGPT subscription (Codex backend ``reasoning.effort``) ── +# The translator always sends a reasoning block and drops output caps, so +# output_tokens stays 0. Levels are the catalogue's supported_reasoning_levels +# minus the client-side "ultra" alias +# (https://github.com/openai/codex/blob/main/codex-rs/models-manager/models.json). + +_CODEX_CATALOG = ( + "https://github.com/openai/codex/blob/main/codex-rs/models-manager/models.json" +) +_CODEX_GPT55 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high", "xhigh"), + provider_default="medium", + source=_CODEX_CATALOG, +) +_CODEX_MAX_LADDER_LOW_DEFAULT = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high", "xhigh", "max"), + provider_default="low", + source=_CODEX_CATALOG, +) +_CODEX_MAX_LADDER_MEDIUM_DEFAULT = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high", "xhigh", "max"), + provider_default="medium", + source=_CODEX_CATALOG, +) + +# ── OpenRouter (``reasoning`` object) ── +# Levels: live GET https://openrouter.ai/api/v1/models ``reasoning`` +# (supported_efforts). Claude 4.5 slugs expose no effort selector, only a +# token budget, which must be >= 1024 and strictly below the request +# max_tokens (https://openrouter.ai/docs/guides/best-practices/reasoning-tokens). +# Explicitly sent sampling parameters are forwarded upstream, where thinking +# rejects them, so every OpenRouter row drops temperature. + +_OPENROUTER_MODELS = "https://openrouter.ai/api/v1/models" +_OPENROUTER_CLAUDE_BUDGET = ReasoningRule( + wire=ReasoningWire.OPENROUTER_BUDGET, + levels=BUDGET_RUNGS, + provider_default=None, + source="https://openrouter.ai/docs/guides/best-practices/reasoning-tokens", + budgets=(4_096, 8_192, 12_288, 16_384), + output_tokens=_ANTHROPIC_OUTPUT_TOKENS, + omit_temperature=True, +) +_OPENROUTER_CLAUDE_SONNET_46 = ReasoningRule( + wire=ReasoningWire.OPENROUTER_EFFORT, + levels=("low", "medium", "high", "max"), + provider_default="medium", + source=_OPENROUTER_MODELS, + output_tokens=_ANTHROPIC_OUTPUT_TOKENS, + omit_temperature=True, +) +_OPENROUTER_CLAUDE_OPUS_46 = replace( + _OPENROUTER_CLAUDE_SONNET_46, provider_default="high" +) +_OPENROUTER_OPENAI_GPT5 = ReasoningRule( + wire=ReasoningWire.OPENROUTER_EFFORT, + levels=("low", "medium", "high"), + provider_default="medium", + source=_OPENROUTER_MODELS, + output_tokens=_EFFORT_OUTPUT_TOKENS, + omit_temperature=True, +) +_OPENROUTER_OPENAI_GPT51 = replace(_OPENROUTER_OPENAI_GPT5, provider_default=None) +_OPENROUTER_OPENAI_XHIGH = ReasoningRule( + wire=ReasoningWire.OPENROUTER_EFFORT, + levels=("low", "medium", "high", "xhigh"), + provider_default="medium", + source=_OPENROUTER_MODELS, + output_tokens=_EFFORT_OUTPUT_TOKENS, + omit_temperature=True, +) +_OPENROUTER_OPENAI_MAX = ReasoningRule( + wire=ReasoningWire.OPENROUTER_EFFORT, + levels=("low", "medium", "high", "xhigh", "max"), + provider_default="medium", + source=_OPENROUTER_MODELS, + output_tokens=_EFFORT_OUTPUT_TOKENS, + omit_temperature=True, +) + +# ── Google Gemini (generateContent ``thinkingConfig``) ── +# https://ai.google.dev/gemini-api/docs/generate-content/thinking. 2.5 models +# take a token budget only (thinkingLevel is a 400 on them); 3.x models take +# a lowercase thinkingLevel. Thinking tokens count toward maxOutputTokens +# (65,536 on every model below), so the cap leaves 8,192 tokens of answer on +# top of the chosen budget. "minimal" is excluded: it is near-off, and an +# error on 3.1-pro / 3.7-flash / 3.8-flash. + +_GEMINI_THINKING_DOCS = ( + "https://ai.google.dev/gemini-api/docs/generate-content/thinking" +) +_GEMINI_OUTPUT_TOKENS = 32_768 +_GEMINI_25_PRO = ReasoningRule( + wire=ReasoningWire.GEMINI_BUDGET, + levels=BUDGET_RUNGS, + provider_default=None, # dynamic thinking + source=_GEMINI_THINKING_DOCS, + budgets=(8_192, 16_384, 24_576, 32_768), + output_tokens=24_576 + 8_192, +) +_GEMINI_25_FLASH = ReasoningRule( + wire=ReasoningWire.GEMINI_BUDGET, + levels=BUDGET_RUNGS, + provider_default=None, # dynamic (flash) or no thinking (flash-lite) + source=_GEMINI_THINKING_DOCS, + budgets=(6_144, 12_288, 18_432, 24_576), + output_tokens=18_432 + 8_192, +) +_GEMINI_3_HIGH_DEFAULT = ReasoningRule( + wire=ReasoningWire.GEMINI_LEVEL, + levels=("low", "medium", "high"), + provider_default="high", + source=_GEMINI_THINKING_DOCS, + output_tokens=_GEMINI_OUTPUT_TOKENS, +) +_GEMINI_3_MEDIUM_DEFAULT = replace(_GEMINI_3_HIGH_DEFAULT, provider_default="medium") +_GEMINI_3_MINIMAL_DEFAULT = replace(_GEMINI_3_HIGH_DEFAULT, provider_default=None) + +# ── xAI (Chat Completions ``reasoning_effort``) ── +# https://docs.x.ai/developers/models/. No Grok model accepts "max", and +# "none" is documented only for grok-4.3. grok-4.5 treats "xhigh" as "high", +# so its distinct levels stop at high. + +_XAI_DOCS = "https://docs.x.ai/developers/models" +_XAI_GROK_43 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high", "xhigh"), + provider_default="low", + source=f"{_XAI_DOCS}/grok-4.3", + output_tokens=_EFFORT_OUTPUT_TOKENS, +) +_XAI_GROK_45 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high"), + provider_default="high", + source="https://docs.x.ai/developers/model-capabilities/text/reasoning", + output_tokens=_EFFORT_OUTPUT_TOKENS, +) +_XAI_GROK_46_PLUS = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high", "xhigh"), + provider_default="high", + source="https://docs.x.ai/developers/model-capabilities/text/reasoning", + output_tokens=_EFFORT_OUTPUT_TOKENS, +) + +# ── DeepSeek (``reasoning_effort``) ── +# https://api-docs.deepseek.com/api/create-chat-completion: none/low/high/max +# (minimal maps to low, medium/xhigh to high), default high; max_tokens up to +# 393,216. Thinking mode ignores temperature without an error. + +_DEEPSEEK_V4 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "high", "max"), + provider_default="high", + source="https://api-docs.deepseek.com/api/create-chat-completion", + output_tokens=_EFFORT_OUTPUT_TOKENS, +) + +# ── Z.ai GLM (``reasoning_effort``, standard route) ── +# https://docs.z.ai/api-reference/llm/chat-completion, default max, 128K +# output. glm-5.3 accepts only low/high/max; glm-5.2 maps low/medium to high +# and xhigh to max, so its distinct levels are high and max. + +_GLM_DOCS = "https://docs.z.ai/api-reference/llm/chat-completion" +_GLM_53 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "high", "max"), + provider_default="max", + source=_GLM_DOCS, + output_tokens=_EFFORT_OUTPUT_TOKENS, +) +_GLM_52 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("high", "max"), + provider_default="max", + source=_GLM_DOCS, + output_tokens=_EFFORT_OUTPUT_TOKENS, +) + +# ── Moonshot Kimi (``reasoning_effort``) ── +# https://platform.kimi.ai/docs/guide/kimi-k3-quickstart: low/high/max, +# default max; always thinks. Temperature is fixed (the profile omits it). + +_KIMI_K3 = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "high", "max"), + provider_default="max", + source="https://platform.kimi.ai/docs/guide/kimi-k3-quickstart", + output_tokens=_EFFORT_OUTPUT_TOKENS, +) + +# ── gpt-oss on Groq / Cerebras (``reasoning_effort``) ── +# low/medium/high, default medium; out-of-set values are a 400. Output caps +# are clamped further by each provider profile's max_output_tokens. + +_GPT_OSS_GROQ = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=("low", "medium", "high"), + provider_default="medium", + source="https://console.groq.com/docs/api-reference", + output_tokens=_EFFORT_OUTPUT_TOKENS, +) +_GPT_OSS_CEREBRAS = replace( + _GPT_OSS_GROQ, source="https://inference-docs.cerebras.ai/capabilities/reasoning" +) + +_CLAUDE_DOCS = "https://platform.claude.com/docs/en/build-with-claude/adaptive-thinking" +_CLAUDE_BUDGET_DOCS = ( + "https://platform.claude.com/docs/en/build-with-claude/extended-thinking" +) + +#: Claude 4.6: adaptive thinking is off unless requested; effort has no +#: xhigh; sampling parameters are incompatible with thinking. +_CLAUDE_46 = ReasoningRule( + wire=ReasoningWire.ANTHROPIC_ADAPTIVE, + levels=("low", "medium", "high", "max"), + provider_default="high", + source=_CLAUDE_DOCS, + output_tokens=_ANTHROPIC_OUTPUT_TOKENS, + omit_temperature=True, +) + +#: Claude 4.7 and later: adaptive is the only thinking mode, the full effort +#: ladder exists, and non-default sampling parameters are a 400. +_CLAUDE_47_PLUS = ReasoningRule( + wire=ReasoningWire.ANTHROPIC_ADAPTIVE, + levels=("low", "medium", "high", "xhigh", "max"), + provider_default="high", + source=_CLAUDE_DOCS, + output_tokens=_ANTHROPIC_OUTPUT_TOKENS, + omit_temperature=True, +) + +#: Claude Opus 5.5: same surface, but its effort default is medium. +_CLAUDE_OPUS_55 = ReasoningRule( + wire=ReasoningWire.ANTHROPIC_ADAPTIVE, + levels=("low", "medium", "high", "xhigh", "max"), + provider_default="medium", + source=_CLAUDE_DOCS, + output_tokens=_ANTHROPIC_OUTPUT_TOKENS, + omit_temperature=True, +) + +#: Claude 4.5 generation: manual budget thinking only (adaptive is a 400, +#: and effort is a 400 on Sonnet 4.5 / Haiku 4.5, so it is never sent). +#: budget_tokens must be >= 1024 and < max_tokens. +_CLAUDE_45_BUDGET = ReasoningRule( + wire=ReasoningWire.ANTHROPIC_BUDGET, + levels=BUDGET_RUNGS, + provider_default=None, + source=_CLAUDE_BUDGET_DOCS, + budgets=(4_096, 8_192, 12_288, 16_384), + output_tokens=_ANTHROPIC_OUTPUT_TOKENS, + omit_temperature=True, +) + + +#: Claude on AWS Bedrock Converse takes the same thinking/effort keys through +#: additionalModelRequestFields, with the same per-model rules +#: (https://docs.aws.amazon.com/bedrock/latest/userguide/claude-messages-adaptive-thinking.html, +#: .../claude-messages-extended-thinking.html). Ids and inference-profile +#: prefixes come from each model's AWS model card. +_BEDROCK_DOCS = ( + "https://docs.aws.amazon.com/bedrock/latest/userguide/" + "claude-messages-adaptive-thinking.html" +) +_BEDROCK_CLAUDE_46 = replace(_CLAUDE_46, source=_BEDROCK_DOCS) +_BEDROCK_CLAUDE_47_PLUS = replace(_CLAUDE_47_PLUS, source=_BEDROCK_DOCS) +_BEDROCK_CLAUDE_OPUS_55 = replace(_CLAUDE_OPUS_55, source=_BEDROCK_DOCS) +_BEDROCK_CLAUDE_45_BUDGET = replace( + _CLAUDE_45_BUDGET, + source=( + "https://docs.aws.amazon.com/bedrock/latest/userguide/" + "claude-messages-extended-thinking.html" + ), +) + + +REASONING_RULES: Mapping[str, Mapping[str, ReasoningRule]] = { + "openai": _surface( + _models( + _OPENAI_GPT5, + "gpt-5", + "gpt-5-2025-08-07", + "gpt-5-mini", + "gpt-5-mini-2025-08-07", + "gpt-5-nano", + "gpt-5-nano-2025-08-07", + ), + _models( + _OPENAI_O_SERIES, + "o3", + "o3-2025-04-16", + "o3-mini", + "o3-mini-2025-01-31", + "o4-mini", + "o4-mini-2025-04-16", + ), + _models(_OPENAI_GPT51, "gpt-5.1", "gpt-5.1-2025-11-13"), + _models( + _OPENAI_GPT52_TO_54, + "gpt-5.2", + "gpt-5.2-2025-12-11", + "gpt-5.4", + "gpt-5.4-2026-03-05", + "gpt-5.4-mini", + "gpt-5.4-mini-2026-03-17", + "gpt-5.4-nano", + "gpt-5.4-nano-2026-03-17", + ), + _models(_OPENAI_GPT55, "gpt-5.5", "gpt-5.5-2026-04-23"), + _models( + _OPENAI_GPT56_PLUS, + "gpt-5.6", + "gpt-5.6-sol", + "gpt-5.6-terra", + "gpt-5.6-luna", + "gpt-6-sol", + "gpt-6-luna", + "gpt-6.1-sol", + ), + _models(_OPENAI_GPT6_ASTRA, "gpt-6-astra"), + ), + "openai_subscription": _surface( + _models(_CODEX_GPT55, "gpt-5.5"), + _models( + _CODEX_MAX_LADDER_LOW_DEFAULT, "gpt-5.6-sol", "gpt-6-astra", "gpt-6.1-sol" + ), + _models( + _CODEX_MAX_LADDER_MEDIUM_DEFAULT, + "gpt-5.6-terra", + "gpt-5.6-luna", + "gpt-6-sol", + "gpt-6-luna", + ), + ), + "openrouter": _surface( + _models( + _OPENROUTER_CLAUDE_BUDGET, + "anthropic/claude-sonnet-4.5", + "anthropic/claude-haiku-4.5", + "anthropic/claude-opus-4.5", + ), + _models(_OPENROUTER_CLAUDE_SONNET_46, "anthropic/claude-sonnet-4.6"), + _models(_OPENROUTER_CLAUDE_OPUS_46, "anthropic/claude-opus-4.6"), + _models(_OPENROUTER_OPENAI_GPT5, "openai/gpt-5", "openai/gpt-5-mini"), + _models(_OPENROUTER_OPENAI_GPT51, "openai/gpt-5.1"), + _models( + _OPENROUTER_OPENAI_XHIGH, + "openai/gpt-5.2", + "openai/gpt-5.4", + "openai/gpt-5.5", + ), + _models( + _OPENROUTER_OPENAI_MAX, + "openai/gpt-5.6-sol", + "openai/gpt-5.6-terra", + "openai/gpt-5.6-luna", + "openai/gpt-6-sol", + "openai/gpt-6-luna", + "openai/gpt-6-astra", + "openai/gpt-6.1-sol", + ), + ), + "gemini": _surface( + _models(_GEMINI_25_PRO, "gemini-2.5-pro"), + _models(_GEMINI_25_FLASH, "gemini-2.5-flash", "gemini-2.5-flash-lite"), + _models( + _GEMINI_3_HIGH_DEFAULT, + "gemini-3-flash-preview", + "gemini-3.1-pro-preview", + "gemini-3.1-pro-preview-customtools", + ), + _models( + _GEMINI_3_MEDIUM_DEFAULT, + "gemini-3.5-flash", + "gemini-3.6-flash", + "gemini-3.7-flash", + "gemini-3.8-flash", + ), + _models( + _GEMINI_3_MINIMAL_DEFAULT, "gemini-3.1-flash-lite", "gemini-3.5-flash-lite" + ), + ), + "grok": _surface( + _models(_XAI_GROK_43, "grok-4.3", "grok-4.3-latest"), + _models(_XAI_GROK_45, "grok-4.5", "grok-4.5-latest"), + _models(_XAI_GROK_46_PLUS, "grok-4.6", "grok-4.7"), + ), + "deepseek": _surface( + _models( + _DEEPSEEK_V4, + "deepseek-flash", + "deepseek-v4-pro", + "deepseek-v4-flash", + "deepseek-v4-flash-vision-exp", + ), + ), + "glm": _surface( + _models(_GLM_53, "glm-5.3", "glm-5.3-flash"), + _models(_GLM_52, "glm-5.2"), + ), + "moonshot": _surface( + _models(_KIMI_K3, "kimi-k3"), + ), + "groq": _surface( + _models( + _GPT_OSS_GROQ, + "openai/gpt-oss-120b", + "openai/gpt-oss-20b", + "openai/gpt-oss-safeguard-20b", + ), + ), + "cerebras": _surface( + _models(_GPT_OSS_CEREBRAS, "gpt-oss-120b"), + ), + "anthropic": _surface( + _models(_CLAUDE_46, "claude-sonnet-4-6", "claude-opus-4-6"), + _models( + _CLAUDE_47_PLUS, + "claude-opus-4-7", + "claude-opus-4-8", + "claude-sonnet-5", + "claude-sonnet-5-5", + "claude-opus-5", + "claude-fable-5", + "claude-fable-5-1", + "claude-mythos-5", + "claude-mythos-5-1", + ), + _models(_CLAUDE_OPUS_55, "claude-opus-5-5"), + _models( + _CLAUDE_45_BUDGET, + "claude-haiku-4-5", + "claude-haiku-4-5-20251001", + "claude-sonnet-4-5", + "claude-sonnet-4-5-20250929", + "claude-opus-4-5", + "claude-opus-4-5-20251101", + ), + ), + "bedrock": _surface( + _models( + _BEDROCK_CLAUDE_45_BUDGET, + *_bedrock_ids( + "anthropic.claude-haiku-4-5-20251001-v1:0", + "us", + "eu", + "au", + "jp", + "in", + "global", + ), + *_bedrock_ids( + "anthropic.claude-sonnet-4-5-20250929-v1:0", + "us", + "eu", + "au", + "jp", + "global", + ), + *_bedrock_ids( + "anthropic.claude-opus-4-5-20251101-v1:0", "us", "eu", "global" + ), + ), + _models( + _BEDROCK_CLAUDE_46, + *_bedrock_ids( + "anthropic.claude-sonnet-4-6", "us", "eu", "au", "jp", "global" + ), + *_bedrock_ids("anthropic.claude-opus-4-6-v1", "us", "eu", "au", "global"), + ), + _models( + _BEDROCK_CLAUDE_47_PLUS, + *_bedrock_ids( + "anthropic.claude-opus-4-7", "us", "eu", "jp", "au", "global" + ), + *_bedrock_ids( + "anthropic.claude-opus-4-8", "us", "eu", "jp", "au", "global" + ), + *_bedrock_ids( + "anthropic.claude-sonnet-5", "us", "eu", "au", "in", "global" + ), + *_bedrock_ids("anthropic.claude-opus-5", "us", "eu", "au", "in", "global"), + *_bedrock_ids("anthropic.claude-fable-5", "us", "global"), + *_bedrock_ids("anthropic.claude-fable-5-1", "us", "global"), + ), + _models( + _BEDROCK_CLAUDE_OPUS_55, + *_bedrock_ids( + "anthropic.claude-opus-5-5", "us", "eu", "au", "jp", "global" + ), + ), + ), +} diff --git a/scripts/probe_reasoning.py b/scripts/probe_reasoning.py new file mode 100644 index 00000000..6588f174 --- /dev/null +++ b/scripts/probe_reasoning.py @@ -0,0 +1,227 @@ +# -*- coding: utf-8 -*- +"""Live check that per-model reasoning defaults are accepted by the providers. + +The unit and golden tests prove what CraftBot SENDS; only the provider can +say whether it ACCEPTS it. This script sends one tiny JSON request per model +through the real LLMInterface (same transports, same output cap as the app) +and reports, per model, the reasoning default that was applied and whether +the provider answered or rejected it. + +Every probe is a real, billed API call (a few hundred tokens each, more for +models that think). Nothing is sent with --dry-run. + +Usage (from the repository root): + python scripts/probe_reasoning.py # the configured LLM model + python scripts/probe_reasoning.py --model openai/gpt-5.2 --model anthropic/claude-sonnet-4-6 + python scripts/probe_reasoning.py --all-rows # every table row with credentials + python scripts/probe_reasoning.py --all-rows --dry-run + +Exit status is 1 when any probe was rejected, else 0. +""" + +from __future__ import annotations + +import argparse +import sys +import time +from dataclasses import dataclass +from pathlib import Path +from typing import List, Optional, Tuple + +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +#: Output cap of the app's LLMInterface (app/llm/interface.py), so probes +#: exercise the same reasoning output-cap logic production does. +APP_MAX_TOKENS = 8_000 + +PROBE_SYSTEM_PROMPT = ( + "You are a connectivity probe. Reply with a single JSON object and nothing else." +) +PROBE_USER_PROMPT = 'Return exactly this JSON object: {"ok": true}' + +#: Table surface -> the provider whose interface serves it. +SUBSCRIPTION_SURFACE = "openai_subscription" + + +@dataclass +class ProbeResult: + target: str + route: str + decision: str + status: str + detail: str + + +def _parse_target(value: str) -> Tuple[str, str]: + provider, sep, model = value.partition("/") + if not sep or not provider or not model: + raise argparse.ArgumentTypeError( + f"{value!r} is not / (for example openai/gpt-5.2)" + ) + return provider, model + + +def _configured_target() -> Tuple[str, str]: + from app.config import get_llm_model, get_llm_provider + + return get_llm_provider(), get_llm_model() + + +def _all_row_targets() -> List[Tuple[str, str, str]]: + """(surface, provider, model) for every row of the reasoning table.""" + from agent_core.core.models.reasoning import REASONING_RULES + + targets = [] + for surface, rows in REASONING_RULES.items(): + provider = "openai" if surface == SUBSCRIPTION_SURFACE else surface + for model in rows: + targets.append((surface, provider, model)) + return targets + + +def _has_aws_credentials() -> bool: + """Whether Bedrock would authenticate: settings keys or boto3's chain. + + boto3 builds a client without credentials and only fails on the first + call, so a missing credential must be detected up front or it would be + reported as the provider rejecting the reasoning parameters. + """ + from app.config import get_aws_credentials + + creds = get_aws_credentials() + if creds.get("access_key_id") and creds.get("secret_access_key"): + return True + import boto3 + + return boto3.Session().get_credentials() is not None + + +def _build_interface(provider: str, model: str): + """A standalone LLMInterface with the app's credentials. + + Returns ``(interface, None)``, or ``(None, reason)`` when the provider + cannot be reached with the configured credentials. + """ + from agent_core.core.impl.llm.interface import LLMInterface + from agent_core.core.models.registry import get_registry + from app.config import get_api_key, get_base_url + + profile = get_registry().get(provider) + if profile is None: + return None, f"unknown provider {provider!r}" + if profile.aws_credential_block and not _has_aws_credentials(): + return None, "no AWS credentials configured" + try: + iface = LLMInterface( + provider=provider, + model=model, + api_key=get_api_key(provider) or None, + base_url=get_base_url(provider) or None, + max_tokens=APP_MAX_TOKENS, + ) + except Exception as exc: # each provider signals missing setup its own way + return None, f"not configured: {exc}" + if not iface.is_initialized: + return None, "not configured" + return iface, None + + +def _probe( + provider: str, model: str, surface: Optional[str], dry_run: bool +) -> Optional[ProbeResult]: + target = f"{provider}/{model}" + iface, unavailable = _build_interface(provider, model) + if iface is None: + return ProbeResult(target, "-", "-", "SKIPPED", unavailable[:200]) + + from agent_core.core.models.reasoning import reasoning_surface + + route = iface._auth_mode + if surface is not None and reasoning_surface(provider, route) != surface: + # This row belongs to the other OpenAI route (API key vs ChatGPT + # subscription), which the current login cannot reach. + return None + if iface.model != model: + return ProbeResult( + target, + route, + "-", + "SKIPPED", + f"provider substitutes {iface.model!r} for this model", + ) + + decision = iface.reasoning_decision() + described = decision.describe() if decision is not None else "no rule (not sent)" + if dry_run: + return ProbeResult(target, route, described, "DRY-RUN", "nothing sent") + + started = time.perf_counter() + try: + reply = iface.generate_response( + system_prompt=PROBE_SYSTEM_PROMPT, + user_prompt=PROBE_USER_PROMPT, + log_response=False, + prompt_name="reasoning_probe", + ) + except Exception as exc: # the provider's rejection is the result + return ProbeResult(target, route, described, "REJECTED", str(exc)[:300]) + elapsed = time.perf_counter() - started + return ProbeResult( + target, route, described, "OK", f"{elapsed:.1f}s: {reply.strip()[:80]}" + ) + + +def main(argv: Optional[List[str]] = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + parser.add_argument( + "--model", + dest="targets", + action="append", + type=_parse_target, + default=[], + metavar="PROVIDER/MODEL", + help="probe this model (repeatable); default is the configured LLM model", + ) + parser.add_argument( + "--all-rows", + action="store_true", + help="probe every reasoning-table row whose provider has credentials", + ) + parser.add_argument( + "--dry-run", + action="store_true", + help="print the reasoning default per model without calling any API", + ) + args = parser.parse_args(argv) + + if args.all_rows: + plan = [(s, p, m) for s, p, m in _all_row_targets()] + elif args.targets: + plan = [(None, p, m) for p, m in args.targets] + else: + provider, model = _configured_target() + plan = [(None, provider, model)] + + results: List[ProbeResult] = [] + for surface, provider, model in plan: + result = _probe(provider, model, surface, args.dry_run) + if result is None: + continue + results.append(result) + print( + f"{result.status:9s} {result.target:55s} [{result.route}] " + f"{result.decision} :: {result.detail}", + flush=True, + ) + + rejected = [r for r in results if r.status == "REJECTED"] + ok = sum(r.status == "OK" for r in results) + skipped = sum(r.status == "SKIPPED" for r in results) + print(f"\n{ok} ok, {len(rejected)} rejected, {skipped} skipped") + return 1 if rejected else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/llm/golden/conftest.py b/tests/llm/golden/conftest.py new file mode 100644 index 00000000..90425d9a --- /dev/null +++ b/tests/llm/golden/conftest.py @@ -0,0 +1,337 @@ +# -*- coding: utf-8 -*- +"""Golden payload harness (Phase 0 of docs/PROVIDER_LAYER_CATCHUP.md). + +Freezes the EXACT request payloads LLMInterface sends per provider so the +Phase 1/2 refactors (registry consolidation, transport extraction) are +provably behavior-preserving. This is the enforcement mechanism for NFR-3 +(KV caching): cache_control placement, cachePoint position, prompt_cache_key +stability, previous_response_id chaining, and session-buffer growth are all +captured in committed JSON snapshots under tests/llm/golden/snapshots/. + +Mocking strategy: +- ``ModelFactory.create`` is patched to return a context dict with recording + fake clients, so LLMInterface's own logic (message building, cache markers, + session buffers) runs unmodified. +- BytePlus uses the REAL BytePlusCacheManager with only its network + chokepoint (``_call_responses_api``) patched, so the previous_response_id + chaining and caching-flag semantics are exercised and frozen too. +- Ollama patches the ``requests`` module reference inside interface.py. + +Snapshot policy: a missing snapshot is created and the test passes (commit +the file). An existing snapshot MUST match; drift fails the test. To +intentionally regenerate after a reviewed behavior change, run with +``GOLDEN_UPDATE=1``. +""" + +from __future__ import annotations + +import copy +import json +import os +from pathlib import Path +from types import SimpleNamespace +from typing import Any, Dict, List + +import pytest + +SNAP_DIR = Path(__file__).resolve().parent / "snapshots" + +# ≥ 500 chars (CacheConfig.min_cache_tokens) so every caching branch fires. +GOLDEN_SYSTEM_PROMPT = ( + "You are CraftBot, a personal always-online agent. " + "You reason step by step, act through a JSON action protocol, and reply " + "with a single JSON object. " +) * 8 + +GOLDEN_TASK_ID = "task-golden" +GOLDEN_CALL_TYPE = "reasoning" +GOLDEN_SESSION_KEY = f"{GOLDEN_TASK_ID}:{GOLDEN_CALL_TYPE}" + + +class CallRecorder: + """Records every provider-bound call (client label, method, payload).""" + + def __init__(self) -> None: + self.calls: List[Dict[str, Any]] = [] + self._turn = 0 + + def next_content(self) -> str: + self._turn += 1 + return json.dumps({"turn": self._turn}) + + def record(self, client: str, method: str, payload: Dict[str, Any]) -> None: + self.calls.append( + {"client": client, "method": method, "payload": copy.deepcopy(payload)} + ) + + +# ─────────────────────────── fake clients ─────────────────────────── + + +class FakeOpenAIClient: + """Stands in for openai.OpenAI — records chat.completions.create kwargs.""" + + def __init__(self, recorder: CallRecorder) -> None: + self._recorder = recorder + self.chat = SimpleNamespace( + completions=SimpleNamespace(create=self._create) + ) + + def _create(self, **kwargs): + self._recorder.record("openai_compat", "chat.completions.create", kwargs) + return SimpleNamespace( + choices=[ + SimpleNamespace( + message=SimpleNamespace(content=self._recorder.next_content()) + ) + ], + usage=SimpleNamespace( + prompt_tokens=100, + completion_tokens=10, + prompt_tokens_details=SimpleNamespace(cached_tokens=0), + prompt_cache_hit_tokens=0, + ), + ) + + +class FakeAnthropicClient: + def __init__(self, recorder: CallRecorder) -> None: + self._recorder = recorder + self.messages = SimpleNamespace(create=self._create) + + def _create(self, **kwargs): + self._recorder.record("anthropic", "messages.create", kwargs) + return SimpleNamespace( + content=[SimpleNamespace(type="text", text=self._recorder.next_content())], + usage=SimpleNamespace( + input_tokens=100, + output_tokens=10, + cache_creation_input_tokens=0, + cache_read_input_tokens=0, + ), + ) + + +class FakeBedrockClient: + def __init__(self, recorder: CallRecorder) -> None: + self._recorder = recorder + + def converse(self, **kwargs): + self._recorder.record("bedrock", "converse", kwargs) + return { + "output": { + "message": { + "content": [{"text": self._recorder.next_content()}] + } + }, + "usage": {"inputTokens": 100, "outputTokens": 10}, + } + + +def make_recording_gemini_client(recorder: CallRecorder): + """A REAL GeminiClient whose single HTTP chokepoint is a recorder. + + Recording the JSON body the client would POST (rather than the Python + kwargs the transport passes it) freezes exactly what reaches the Gemini + API, including the generationConfig the client assembles. + """ + from agent_core.core.llm.google_gemini_client import GeminiClient + + client = GeminiClient(api_key="test-gemini-key") + + def _post_json(path: str, payload: Dict[str, Any]) -> Dict[str, Any]: + recorder.record("gemini", "generateContent", {"path": path, "body": payload}) + return { + "candidates": [ + { + "content": {"parts": [{"text": recorder.next_content()}]}, + "finishReason": "STOP", + } + ], + "usageMetadata": { + "totalTokenCount": 110, + "promptTokenCount": 100, + "candidatesTokenCount": 10, + "cachedContentTokenCount": 0, + }, + } + + client._post_json = _post_json + return client + + +def make_fake_requests(recorder: CallRecorder): + """Fake for the module-level ``requests`` used by _generate_ollama.""" + + def post(url, json=None, timeout=None): # noqa: A002 - mirrors requests API + recorder.record("ollama", "requests.post", {"url": url, "json": json}) + return SimpleNamespace( + raise_for_status=lambda: None, + json=lambda: { + "response": recorder.next_content(), + "prompt_eval_count": 100, + "eval_count": 10, + }, + ) + + return SimpleNamespace(post=post) + + +def make_fake_byteplus_transport(recorder: CallRecorder): + """Replacement for BytePlusCacheManager._call_responses_api. + + The REAL manager logic (session registry, previous_response_id chaining, + caching flags) runs; only the HTTP hop is faked. Payload keys mirror the + method's signature so the chaining semantics land in the snapshot. + """ + counter = {"n": 0} + + def _call_responses_api( + self, + input_messages, + temperature, + max_tokens, + previous_response_id=None, + caching_enabled=True, + caching_prefix=False, + ): + recorder.record( + "byteplus", + "_call_responses_api", + { + "input_messages": input_messages, + "temperature": temperature, + "max_tokens": max_tokens, + "previous_response_id": previous_response_id, + "caching_enabled": caching_enabled, + "caching_prefix": caching_prefix, + }, + ) + counter["n"] += 1 + return { + "id": f"resp_{counter['n']}", + "output": [ + { + "type": "message", + "role": "assistant", + "content": [ + {"type": "output_text", "text": recorder.next_content()} + ], + } + ], + "usage": { + "input_tokens": 100, + "output_tokens": 10, + "total_tokens": 110, + "input_tokens_details": {"cached_tokens": 0}, + }, + } + + return _call_responses_api + + +# ─────────────────────────── fixture ─────────────────────────── + + +def build_interface(monkeypatch, provider: str, model: str): + """Construct an LLMInterface for ``provider`` with recording fakes.""" + from agent_core.core.models.factory import ModelFactory + import agent_core.core.impl.llm.interface as interface_mod + from agent_core.core.impl.llm.cache.byteplus import BytePlusCacheManager + + recorder = CallRecorder() + + ctx: Dict[str, Any] = { + "provider": provider, + "model": model, + "client": None, + "gemini_client": None, + "anthropic_client": None, + "bedrock_client": None, + "remote_url": None, + "byteplus": None, + "initialized": True, + "auth_mode": "api_key", + } + + if provider == "anthropic": + ctx["anthropic_client"] = FakeAnthropicClient(recorder) + elif provider == "bedrock": + ctx["bedrock_client"] = FakeBedrockClient(recorder) + elif provider == "gemini": + ctx["gemini_client"] = make_recording_gemini_client(recorder) + elif provider == "byteplus": + ctx["byteplus"] = { + "api_key": "test-byteplus-key", + "base_url": "https://fake.byteplus.test/api/v3", + } + monkeypatch.setattr( + BytePlusCacheManager, + "_call_responses_api", + make_fake_byteplus_transport(recorder), + ) + elif provider == "remote": + ctx["remote_url"] = "http://localhost:11434" + # Ollama's HTTP hop lives in the chat_completions transport since + # Phase 2; patch the module reference the live code actually uses. + from agent_core.core.impl.llm.transports import chat_completions + + monkeypatch.setattr( + chat_completions, "requests", make_fake_requests(recorder) + ) + else: + # openai / deepseek / grok / openrouter / glm / fugu / minimax / moonshot + ctx["client"] = FakeOpenAIClient(recorder) + + def fake_create(**kwargs): + return dict(ctx) + + monkeypatch.setattr(ModelFactory, "create", staticmethod(fake_create)) + + iface = interface_mod.LLMInterface(provider=provider, model=model) + return iface, recorder + + +def collect_buffers(iface) -> Dict[str, Any]: + """Deep-copy the accumulated session histories (NFR-3 state). + + One interface serves one provider, so its single ``_session_histories`` + map holds whatever message shape that provider's session branch builds. + """ + return {"session_histories": copy.deepcopy(iface._session_histories)} + + +# ─────────────────────────── snapshots ─────────────────────────── + + +def assert_snapshot(name: str, data: Dict[str, Any]) -> None: + SNAP_DIR.mkdir(parents=True, exist_ok=True) + path = SNAP_DIR / f"{name}.json" + canonical = json.loads(json.dumps(data)) # normalize tuples etc. + + if os.environ.get("GOLDEN_UPDATE") == "1" or not path.exists(): + path.write_text( + json.dumps(canonical, indent=2, sort_keys=True, ensure_ascii=False) + + "\n", + encoding="utf-8", + ) + return + + expected = json.loads(path.read_text(encoding="utf-8")) + assert canonical == expected, ( + f"Golden payload drift for {name!r}.\n" + f"A request payload, cache marker, or session buffer changed. This is a " + f"behavior change by definition (see docs/PROVIDER_LAYER_CATCHUP.md " + f"section 12.3). If intentional and reviewed, regenerate with " + f"GOLDEN_UPDATE=1 and commit the diff." + ) + + +@pytest.fixture() +def golden(monkeypatch): + """Factory fixture: golden(provider, model) -> (iface, recorder).""" + + def _build(provider: str, model: str): + return build_interface(monkeypatch, provider, model) + + return _build diff --git a/tests/llm/golden/snapshots/anthropic.json b/tests/llm/golden/snapshots/anthropic.json new file mode 100644 index 00000000..e419f5a0 --- /dev/null +++ b/tests/llm/golden/snapshots/anthropic.json @@ -0,0 +1,195 @@ +{ + "calls": [ + { + "client": "anthropic", + "method": "messages.create", + "payload": { + "max_tokens": 21000, + "messages": [ + { + "content": "sessionless turn", + "role": "user" + } + ], + "model": "claude-sonnet-4-6", + "output_config": { + "effort": "high" + }, + "system": [ + { + "cache_control": { + "type": "ephemeral" + }, + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "type": "text" + } + ], + "thinking": { + "type": "adaptive" + } + } + }, + { + "client": "anthropic", + "method": "messages.create", + "payload": { + "max_tokens": 21000, + "messages": [ + { + "content": "session turn 1", + "role": "user" + } + ], + "model": "claude-sonnet-4-6", + "output_config": { + "effort": "high" + }, + "system": [ + { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "type": "text" + } + ], + "thinking": { + "type": "adaptive" + } + } + }, + { + "client": "anthropic", + "method": "messages.create", + "payload": { + "max_tokens": 21000, + "messages": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": [ + { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "text": "{\"turn\": 2}", + "type": "text" + } + ], + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + } + ], + "model": "claude-sonnet-4-6", + "output_config": { + "effort": "high" + }, + "system": [ + { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "type": "text" + } + ], + "thinking": { + "type": "adaptive" + } + } + }, + { + "client": "anthropic", + "method": "messages.create", + "payload": { + "max_tokens": 21000, + "messages": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": [ + { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "text": "{\"turn\": 3}", + "type": "text" + } + ], + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + } + ], + "model": "claude-sonnet-4-6", + "output_config": { + "effort": "high" + }, + "system": [ + { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "type": "text" + } + ], + "thinking": { + "type": "adaptive" + } + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + }, + { + "content": "{\"turn\": 4}", + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/bedrock.json b/tests/llm/golden/snapshots/bedrock.json new file mode 100644 index 00000000..adf6686e --- /dev/null +++ b/tests/llm/golden/snapshots/bedrock.json @@ -0,0 +1,245 @@ +{ + "calls": [ + { + "client": "bedrock", + "method": "converse", + "payload": { + "additionalModelRequestFields": { + "thinking": { + "budget_tokens": 12288, + "type": "enabled" + } + }, + "inferenceConfig": { + "maxTokens": 50000 + }, + "messages": [ + { + "content": [ + { + "text": "sessionless turn" + } + ], + "role": "user" + } + ], + "modelId": "us.anthropic.claude-haiku-4-5-20251001-v1:0", + "system": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + { + "client": "bedrock", + "method": "converse", + "payload": { + "additionalModelRequestFields": { + "thinking": { + "budget_tokens": 12288, + "type": "enabled" + } + }, + "inferenceConfig": { + "maxTokens": 50000 + }, + "messages": [ + { + "content": [ + { + "text": "session turn 1" + } + ], + "role": "user" + } + ], + "modelId": "us.anthropic.claude-haiku-4-5-20251001-v1:0", + "system": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + }, + { + "cachePoint": { + "type": "default" + } + } + ] + } + }, + { + "client": "bedrock", + "method": "converse", + "payload": { + "additionalModelRequestFields": { + "thinking": { + "budget_tokens": 12288, + "type": "enabled" + } + }, + "inferenceConfig": { + "maxTokens": 50000 + }, + "messages": [ + { + "content": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 2}" + }, + { + "cachePoint": { + "type": "default" + } + } + ], + "role": "assistant" + }, + { + "content": [ + { + "text": "session turn 2" + } + ], + "role": "user" + } + ], + "modelId": "us.anthropic.claude-haiku-4-5-20251001-v1:0", + "system": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + { + "client": "bedrock", + "method": "converse", + "payload": { + "additionalModelRequestFields": { + "thinking": { + "budget_tokens": 12288, + "type": "enabled" + } + }, + "inferenceConfig": { + "maxTokens": 50000 + }, + "messages": [ + { + "content": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 2}" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "text": "session turn 2" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 3}" + }, + { + "cachePoint": { + "type": "default" + } + } + ], + "role": "assistant" + }, + { + "content": [ + { + "text": "session turn 3" + } + ], + "role": "user" + } + ], + "modelId": "us.anthropic.claude-haiku-4-5-20251001-v1:0", + "system": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 2}" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "text": "session turn 2" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 3}" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "text": "session turn 3" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 4}" + } + ], + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/byteplus.json b/tests/llm/golden/snapshots/byteplus.json new file mode 100644 index 00000000..68f6914b --- /dev/null +++ b/tests/llm/golden/snapshots/byteplus.json @@ -0,0 +1,83 @@ +{ + "calls": [ + { + "client": "byteplus", + "method": "_call_responses_api", + "payload": { + "caching_enabled": true, + "caching_prefix": false, + "input_messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "sessionless turn", + "role": "user" + } + ], + "max_tokens": 50000, + "previous_response_id": null, + "temperature": 0.0 + } + }, + { + "client": "byteplus", + "method": "_call_responses_api", + "payload": { + "caching_enabled": true, + "caching_prefix": false, + "input_messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + } + ], + "max_tokens": 50000, + "previous_response_id": null, + "temperature": 0.0 + } + }, + { + "client": "byteplus", + "method": "_call_responses_api", + "payload": { + "caching_enabled": true, + "caching_prefix": false, + "input_messages": [ + { + "content": "session turn 2", + "role": "user" + } + ], + "max_tokens": 50000, + "previous_response_id": "resp_2", + "temperature": 0.0 + } + }, + { + "client": "byteplus", + "method": "_call_responses_api", + "payload": { + "caching_enabled": true, + "caching_prefix": false, + "input_messages": [ + { + "content": "session turn 3", + "role": "user" + } + ], + "max_tokens": 50000, + "previous_response_id": "resp_3", + "temperature": 0.0 + } + } + ], + "session_buffers": { + "session_histories": {} + } +} diff --git a/tests/llm/golden/snapshots/deepseek.json b/tests/llm/golden/snapshots/deepseek.json new file mode 100644 index 00000000..9d481058 --- /dev/null +++ b/tests/llm/golden/snapshots/deepseek.json @@ -0,0 +1,158 @@ +{ + "calls": [ + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "sessionless turn", + "role": "user" + } + ], + "model": "deepseek-chat", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + } + ], + "model": "deepseek-chat", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + } + ], + "model": "deepseek-chat", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + } + ], + "model": "deepseek-chat", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + }, + { + "content": "{\"turn\": 4}", + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/gemini.json b/tests/llm/golden/snapshots/gemini.json new file mode 100644 index 00000000..cadf0be7 --- /dev/null +++ b/tests/llm/golden/snapshots/gemini.json @@ -0,0 +1,242 @@ +{ + "calls": [ + { + "client": "gemini", + "method": "generateContent", + "payload": { + "body": { + "contents": [ + { + "parts": [ + { + "text": "sessionless turn" + } + ], + "role": "user" + } + ], + "generationConfig": { + "maxOutputTokens": 50000, + "responseMimeType": "application/json", + "temperature": 0.0, + "thinkingConfig": { + "thinkingBudget": 24576 + } + }, + "systemInstruction": { + "parts": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + "path": "models/gemini-2.5-pro:generateContent" + } + }, + { + "client": "gemini", + "method": "generateContent", + "payload": { + "body": { + "contents": [ + { + "parts": [ + { + "text": "session turn 1" + } + ], + "role": "user" + } + ], + "generationConfig": { + "maxOutputTokens": 50000, + "responseMimeType": "application/json", + "temperature": 0.0, + "thinkingConfig": { + "thinkingBudget": 24576 + } + }, + "systemInstruction": { + "parts": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + "path": "models/gemini-2.5-pro:generateContent" + } + }, + { + "client": "gemini", + "method": "generateContent", + "payload": { + "body": { + "contents": [ + { + "parts": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 2}" + } + ], + "role": "model" + }, + { + "parts": [ + { + "text": "session turn 2" + } + ], + "role": "user" + } + ], + "generationConfig": { + "maxOutputTokens": 50000, + "responseMimeType": "application/json", + "temperature": 0.0, + "thinkingConfig": { + "thinkingBudget": 24576 + } + }, + "systemInstruction": { + "parts": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + "path": "models/gemini-2.5-pro:generateContent" + } + }, + { + "client": "gemini", + "method": "generateContent", + "payload": { + "body": { + "contents": [ + { + "parts": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 2}" + } + ], + "role": "model" + }, + { + "parts": [ + { + "text": "session turn 2" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 3}" + } + ], + "role": "model" + }, + { + "parts": [ + { + "text": "session turn 3" + } + ], + "role": "user" + } + ], + "generationConfig": { + "maxOutputTokens": 50000, + "responseMimeType": "application/json", + "temperature": 0.0, + "thinkingConfig": { + "thinkingBudget": 24576 + } + }, + "systemInstruction": { + "parts": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + "path": "models/gemini-2.5-pro:generateContent" + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "parts": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 2}" + } + ], + "role": "model" + }, + { + "parts": [ + { + "text": "session turn 2" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 3}" + } + ], + "role": "model" + }, + { + "parts": [ + { + "text": "session turn 3" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 4}" + } + ], + "role": "model" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/groq.json b/tests/llm/golden/snapshots/groq.json new file mode 100644 index 00000000..470d2703 --- /dev/null +++ b/tests/llm/golden/snapshots/groq.json @@ -0,0 +1,146 @@ +{ + "calls": [ + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "max_completion_tokens": 32768, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "sessionless turn", + "role": "user" + } + ], + "model": "llama-3.3-70b-versatile", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "max_completion_tokens": 32768, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + } + ], + "model": "llama-3.3-70b-versatile", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "max_completion_tokens": 32768, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + } + ], + "model": "llama-3.3-70b-versatile", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "max_completion_tokens": 32768, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + } + ], + "model": "llama-3.3-70b-versatile", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + }, + { + "content": "{\"turn\": 4}", + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/openai.json b/tests/llm/golden/snapshots/openai.json index 3a10ea09..e992d246 100644 --- a/tests/llm/golden/snapshots/openai.json +++ b/tests/llm/golden/snapshots/openai.json @@ -19,6 +19,7 @@ } ], "model": "gpt-5.2-2025-12-11", + "reasoning_effort": "high", "response_format": { "type": "json_object" } @@ -43,6 +44,7 @@ } ], "model": "gpt-5.2-2025-12-11", + "reasoning_effort": "high", "response_format": { "type": "json_object" } @@ -75,6 +77,7 @@ } ], "model": "gpt-5.2-2025-12-11", + "reasoning_effort": "high", "response_format": { "type": "json_object" } @@ -115,6 +118,7 @@ } ], "model": "gpt-5.2-2025-12-11", + "reasoning_effort": "high", "response_format": { "type": "json_object" } @@ -122,10 +126,7 @@ } ], "session_buffers": { - "anthropic": {}, - "bedrock": {}, - "gemini": {}, - "openai_compat": { + "session_histories": { "task-golden:reasoning": [ { "content": "session turn 1", @@ -152,7 +153,6 @@ "role": "assistant" } ] - }, - "openrouter_anthropic": {} + } } } diff --git a/tests/llm/golden/snapshots/openrouter_claude.json b/tests/llm/golden/snapshots/openrouter_claude.json new file mode 100644 index 00000000..e31a4903 --- /dev/null +++ b/tests/llm/golden/snapshots/openrouter_claude.json @@ -0,0 +1,181 @@ +{ + "calls": [ + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "cache_control": { + "type": "ephemeral" + }, + "prompt_cache_key": "907eed6c3582215b", + "reasoning": { + "max_tokens": 12288 + } + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "sessionless turn", + "role": "user" + } + ], + "model": "anthropic/claude-sonnet-4.5", + "response_format": { + "type": "json_object" + } + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "prompt_cache_key": "reasoning_907eed6c3582215b", + "reasoning": { + "max_tokens": 12288 + } + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + } + ], + "model": "anthropic/claude-sonnet-4.5", + "response_format": { + "type": "json_object" + } + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "prompt_cache_key": "reasoning_907eed6c3582215b", + "reasoning": { + "max_tokens": 12288 + } + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + } + ], + "model": "anthropic/claude-sonnet-4.5", + "response_format": { + "type": "json_object" + } + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "prompt_cache_key": "reasoning_907eed6c3582215b", + "reasoning": { + "max_tokens": 12288 + } + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + } + ], + "model": "anthropic/claude-sonnet-4.5", + "response_format": { + "type": "json_object" + } + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + }, + { + "content": "{\"turn\": 4}", + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/openrouter_non_claude.json b/tests/llm/golden/snapshots/openrouter_non_claude.json new file mode 100644 index 00000000..0665ccc5 --- /dev/null +++ b/tests/llm/golden/snapshots/openrouter_non_claude.json @@ -0,0 +1,158 @@ +{ + "calls": [ + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "sessionless turn", + "role": "user" + } + ], + "model": "deepseek/deepseek-chat", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + } + ], + "model": "deepseek/deepseek-chat", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + } + ], + "model": "deepseek/deepseek-chat", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + } + ], + "model": "deepseek/deepseek-chat", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + }, + { + "content": "{\"turn\": 4}", + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/remote.json b/tests/llm/golden/snapshots/remote.json new file mode 100644 index 00000000..f2fb10d6 --- /dev/null +++ b/tests/llm/golden/snapshots/remote.json @@ -0,0 +1,72 @@ +{ + "calls": [ + { + "client": "ollama", + "method": "requests.post", + "payload": { + "json": { + "format": "json", + "model": "llama3.2:3b", + "options": { + "temperature": 0.0 + }, + "prompt": "sessionless turn", + "stream": false, + "system": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + }, + "url": "http://localhost:11434/api/generate" + } + }, + { + "client": "ollama", + "method": "requests.post", + "payload": { + "json": { + "format": "json", + "model": "llama3.2:3b", + "options": { + "temperature": 0.0 + }, + "prompt": "session turn 1", + "stream": false + }, + "url": "http://localhost:11434/api/generate" + } + }, + { + "client": "ollama", + "method": "requests.post", + "payload": { + "json": { + "format": "json", + "model": "llama3.2:3b", + "options": { + "temperature": 0.0 + }, + "prompt": "session turn 2", + "stream": false + }, + "url": "http://localhost:11434/api/generate" + } + }, + { + "client": "ollama", + "method": "requests.post", + "payload": { + "json": { + "format": "json", + "model": "llama3.2:3b", + "options": { + "temperature": 0.0 + }, + "prompt": "session turn 3", + "stream": false + }, + "url": "http://localhost:11434/api/generate" + } + } + ], + "session_buffers": { + "session_histories": {} + } +} diff --git a/tests/llm/golden/snapshots/unruled_anthropic.json b/tests/llm/golden/snapshots/unruled_anthropic.json new file mode 100644 index 00000000..e3bdcd17 --- /dev/null +++ b/tests/llm/golden/snapshots/unruled_anthropic.json @@ -0,0 +1,183 @@ +{ + "calls": [ + { + "client": "anthropic", + "method": "messages.create", + "payload": { + "extra_body": { + "temperature": 0.0 + }, + "max_tokens": 16384, + "messages": [ + { + "content": "sessionless turn", + "role": "user" + } + ], + "model": "claude-3-5-haiku-20241022", + "system": [ + { + "cache_control": { + "type": "ephemeral" + }, + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "type": "text" + } + ] + } + }, + { + "client": "anthropic", + "method": "messages.create", + "payload": { + "extra_body": { + "temperature": 0.0 + }, + "max_tokens": 16384, + "messages": [ + { + "content": "session turn 1", + "role": "user" + } + ], + "model": "claude-3-5-haiku-20241022", + "system": [ + { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "type": "text" + } + ] + } + }, + { + "client": "anthropic", + "method": "messages.create", + "payload": { + "extra_body": { + "temperature": 0.0 + }, + "max_tokens": 16384, + "messages": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": [ + { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "text": "{\"turn\": 2}", + "type": "text" + } + ], + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + } + ], + "model": "claude-3-5-haiku-20241022", + "system": [ + { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "type": "text" + } + ] + } + }, + { + "client": "anthropic", + "method": "messages.create", + "payload": { + "extra_body": { + "temperature": 0.0 + }, + "max_tokens": 16384, + "messages": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": [ + { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "text": "{\"turn\": 3}", + "type": "text" + } + ], + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + } + ], + "model": "claude-3-5-haiku-20241022", + "system": [ + { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "type": "text" + } + ] + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + }, + { + "content": "{\"turn\": 4}", + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/unruled_bedrock.json b/tests/llm/golden/snapshots/unruled_bedrock.json new file mode 100644 index 00000000..30778432 --- /dev/null +++ b/tests/llm/golden/snapshots/unruled_bedrock.json @@ -0,0 +1,220 @@ +{ + "calls": [ + { + "client": "bedrock", + "method": "converse", + "payload": { + "inferenceConfig": { + "maxTokens": 50000, + "temperature": 0.0 + }, + "messages": [ + { + "content": [ + { + "text": "sessionless turn" + } + ], + "role": "user" + } + ], + "modelId": "meta.llama3-3-70b-instruct-v1:0", + "system": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + { + "client": "bedrock", + "method": "converse", + "payload": { + "inferenceConfig": { + "maxTokens": 50000, + "temperature": 0.0 + }, + "messages": [ + { + "content": [ + { + "text": "session turn 1" + } + ], + "role": "user" + } + ], + "modelId": "meta.llama3-3-70b-instruct-v1:0", + "system": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + { + "client": "bedrock", + "method": "converse", + "payload": { + "inferenceConfig": { + "maxTokens": 50000, + "temperature": 0.0 + }, + "messages": [ + { + "content": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 2}" + }, + { + "cachePoint": { + "type": "default" + } + } + ], + "role": "assistant" + }, + { + "content": [ + { + "text": "session turn 2" + } + ], + "role": "user" + } + ], + "modelId": "meta.llama3-3-70b-instruct-v1:0", + "system": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + { + "client": "bedrock", + "method": "converse", + "payload": { + "inferenceConfig": { + "maxTokens": 50000, + "temperature": 0.0 + }, + "messages": [ + { + "content": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 2}" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "text": "session turn 2" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 3}" + }, + { + "cachePoint": { + "type": "default" + } + } + ], + "role": "assistant" + }, + { + "content": [ + { + "text": "session turn 3" + } + ], + "role": "user" + } + ], + "modelId": "meta.llama3-3-70b-instruct-v1:0", + "system": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 2}" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "text": "session turn 2" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 3}" + } + ], + "role": "assistant" + }, + { + "content": [ + { + "text": "session turn 3" + } + ], + "role": "user" + }, + { + "content": [ + { + "text": "{\"turn\": 4}" + } + ], + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/unruled_gemini.json b/tests/llm/golden/snapshots/unruled_gemini.json new file mode 100644 index 00000000..295ac159 --- /dev/null +++ b/tests/llm/golden/snapshots/unruled_gemini.json @@ -0,0 +1,230 @@ +{ + "calls": [ + { + "client": "gemini", + "method": "generateContent", + "payload": { + "body": { + "contents": [ + { + "parts": [ + { + "text": "sessionless turn" + } + ], + "role": "user" + } + ], + "generationConfig": { + "maxOutputTokens": 50000, + "responseMimeType": "application/json", + "temperature": 0.0 + }, + "systemInstruction": { + "parts": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + "path": "models/gemini-2.0-flash:generateContent" + } + }, + { + "client": "gemini", + "method": "generateContent", + "payload": { + "body": { + "contents": [ + { + "parts": [ + { + "text": "session turn 1" + } + ], + "role": "user" + } + ], + "generationConfig": { + "maxOutputTokens": 50000, + "responseMimeType": "application/json", + "temperature": 0.0 + }, + "systemInstruction": { + "parts": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + "path": "models/gemini-2.0-flash:generateContent" + } + }, + { + "client": "gemini", + "method": "generateContent", + "payload": { + "body": { + "contents": [ + { + "parts": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 2}" + } + ], + "role": "model" + }, + { + "parts": [ + { + "text": "session turn 2" + } + ], + "role": "user" + } + ], + "generationConfig": { + "maxOutputTokens": 50000, + "responseMimeType": "application/json", + "temperature": 0.0 + }, + "systemInstruction": { + "parts": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + "path": "models/gemini-2.0-flash:generateContent" + } + }, + { + "client": "gemini", + "method": "generateContent", + "payload": { + "body": { + "contents": [ + { + "parts": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 2}" + } + ], + "role": "model" + }, + { + "parts": [ + { + "text": "session turn 2" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 3}" + } + ], + "role": "model" + }, + { + "parts": [ + { + "text": "session turn 3" + } + ], + "role": "user" + } + ], + "generationConfig": { + "maxOutputTokens": 50000, + "responseMimeType": "application/json", + "temperature": 0.0 + }, + "systemInstruction": { + "parts": [ + { + "text": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. " + } + ] + } + }, + "path": "models/gemini-2.0-flash:generateContent" + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "parts": [ + { + "text": "session turn 1" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 2}" + } + ], + "role": "model" + }, + { + "parts": [ + { + "text": "session turn 2" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 3}" + } + ], + "role": "model" + }, + { + "parts": [ + { + "text": "session turn 3" + } + ], + "role": "user" + }, + { + "parts": [ + { + "text": "{\"turn\": 4}" + } + ], + "role": "model" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/unruled_grok.json b/tests/llm/golden/snapshots/unruled_grok.json new file mode 100644 index 00000000..f679b356 --- /dev/null +++ b/tests/llm/golden/snapshots/unruled_grok.json @@ -0,0 +1,158 @@ +{ + "calls": [ + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "sessionless turn", + "role": "user" + } + ], + "model": "grok-3", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + } + ], + "model": "grok-3", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + } + ], + "model": "grok-3", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + } + ], + "model": "grok-3", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + }, + { + "content": "{\"turn\": 4}", + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/unruled_openai.json b/tests/llm/golden/snapshots/unruled_openai.json new file mode 100644 index 00000000..a0ca596d --- /dev/null +++ b/tests/llm/golden/snapshots/unruled_openai.json @@ -0,0 +1,154 @@ +{ + "calls": [ + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "907eed6c3582215b" + }, + "max_completion_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "sessionless turn", + "role": "user" + } + ], + "model": "gpt-4o", + "response_format": { + "type": "json_object" + } + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_completion_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + } + ], + "model": "gpt-4o", + "response_format": { + "type": "json_object" + } + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_completion_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + } + ], + "model": "gpt-4o", + "response_format": { + "type": "json_object" + } + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_completion_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + } + ], + "model": "gpt-4o", + "response_format": { + "type": "json_object" + } + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + }, + { + "content": "{\"turn\": 4}", + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/snapshots/unruled_openrouter_claude.json b/tests/llm/golden/snapshots/unruled_openrouter_claude.json new file mode 100644 index 00000000..4214b7f7 --- /dev/null +++ b/tests/llm/golden/snapshots/unruled_openrouter_claude.json @@ -0,0 +1,173 @@ +{ + "calls": [ + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "cache_control": { + "type": "ephemeral" + }, + "prompt_cache_key": "907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "sessionless turn", + "role": "user" + } + ], + "model": "anthropic/claude-3.5-sonnet", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + } + ], + "model": "anthropic/claude-3.5-sonnet", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + } + ], + "model": "anthropic/claude-3.5-sonnet", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + }, + { + "client": "openai_compat", + "method": "chat.completions.create", + "payload": { + "extra_body": { + "cache_control": { + "ttl": "1h", + "type": "ephemeral" + }, + "prompt_cache_key": "reasoning_907eed6c3582215b" + }, + "max_tokens": 50000, + "messages": [ + { + "content": "You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. You are CraftBot, a personal always-online agent. You reason step by step, act through a JSON action protocol, and reply with a single JSON object. ", + "role": "system" + }, + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + } + ], + "model": "anthropic/claude-3.5-sonnet", + "response_format": { + "type": "json_object" + }, + "temperature": 0.0 + } + } + ], + "session_buffers": { + "session_histories": { + "task-golden:reasoning": [ + { + "content": "session turn 1", + "role": "user" + }, + { + "content": "{\"turn\": 2}", + "role": "assistant" + }, + { + "content": "session turn 2", + "role": "user" + }, + { + "content": "{\"turn\": 3}", + "role": "assistant" + }, + { + "content": "session turn 3", + "role": "user" + }, + { + "content": "{\"turn\": 4}", + "role": "assistant" + } + ] + } + } +} diff --git a/tests/llm/golden/test_golden_payloads.py b/tests/llm/golden/test_golden_payloads.py index f15b8846..0225e515 100644 --- a/tests/llm/golden/test_golden_payloads.py +++ b/tests/llm/golden/test_golden_payloads.py @@ -12,8 +12,11 @@ from __future__ import annotations +import pytest + from .conftest import ( GOLDEN_CALL_TYPE, + GOLDEN_SESSION_KEY, GOLDEN_SYSTEM_PROMPT, GOLDEN_TASK_ID, assert_snapshot, @@ -94,6 +97,16 @@ def test_anthropic(golden): assert marked[0][0] == 3 # index of the last assistant message assert marked[0][1]["cache_control"] == {"type": "ephemeral", "ttl": "1h"} + # Reasoning default (Claude 4.6 row): adaptive thinking at effort high, + # temperature dropped (incompatible with thinking), max_tokens raised + # for thinking but kept under the SDK's non-streaming ceiling. + for call in creates: + payload = call["payload"] + assert payload["thinking"] == {"type": "adaptive"} + assert payload["output_config"] == {"effort": "high"} + assert "extra_body" not in payload + assert payload["max_tokens"] == 21000 + snapshot_scenario("anthropic", iface, rec) @@ -137,6 +150,16 @@ def test_bedrock(golden): assert i == 3 # last assistant in [u1, a1, u2, a2, new_user] assert j == len(turn3["messages"][i]["content"]) - 1 + # Reasoning default (Claude 4.5 budget row via Converse): budget thinking + # at the "high" rung, temperature dropped, maxTokens never lowered. + for call in converses: + payload = call["payload"] + assert payload["additionalModelRequestFields"] == { + "thinking": {"type": "enabled", "budget_tokens": 12288} + } + assert "temperature" not in payload["inferenceConfig"] + assert payload["inferenceConfig"]["maxTokens"] == 50000 + snapshot_scenario("bedrock", iface, rec) @@ -175,6 +198,11 @@ def test_openai(golden): assert msgs[0]["role"] == "system" assert msgs[-1]["role"] == "user" + # Reasoning default (gpt-5.2 row): one level below xhigh. The output cap + # is never lowered (this fixture's max_tokens already exceeds the row's). + assert all(c["payload"]["reasoning_effort"] == "high" for c in creates) + assert all(c["payload"]["max_completion_tokens"] == 50000 for c in creates) + snapshot_scenario("openai", iface, rec) @@ -330,11 +358,18 @@ def test_openrouter_claude(golden): assert extra["cache_control"] == {"type": "ephemeral", "ttl": "1h"} assert extra["prompt_cache_key"].startswith(f"{GOLDEN_CALL_TYPE}_") - # Accumulation must use the OpenRouter-Anthropic buffer, not openai_compat. - buffers = collect_buffers(iface) - key = f"{GOLDEN_TASK_ID}:{GOLDEN_CALL_TYPE}" - assert len(buffers["openrouter_anthropic"][key]) == 6 # 3 turns x (u, a) - assert buffers["openai_compat"] == {} + # Accumulation: 3 turns x (user, assistant) in OpenAI message shape. + history = collect_buffers(iface)["session_histories"][GOLDEN_SESSION_KEY] + assert [m["role"] for m in history] == ["user", "assistant"] * 3 + + # Reasoning default (Claude 4.5 budget row via OpenRouter): a fixed + # thinking budget beside cache_control, temperature dropped, and the + # request max_tokens strictly above the budget. + for call in creates: + payload = call["payload"] + assert payload["extra_body"]["reasoning"] == {"max_tokens": 12288} + assert "temperature" not in payload + assert payload["max_tokens"] > 12288 snapshot_scenario("openrouter_claude", iface, rec) @@ -351,10 +386,8 @@ def test_openrouter_non_claude(golden): extra = call["payload"].get("extra_body", {}) assert "cache_control" not in extra - buffers = collect_buffers(iface) - key = f"{GOLDEN_TASK_ID}:{GOLDEN_CALL_TYPE}" - assert len(buffers["openai_compat"][key]) == 6 - assert buffers["openrouter_anthropic"] == {} + history = collect_buffers(iface)["session_histories"][GOLDEN_SESSION_KEY] + assert [m["role"] for m in history] == ["user", "assistant"] * 3 snapshot_scenario("openrouter_non_claude", iface, rec) @@ -373,9 +406,8 @@ def test_groq_new_provider(golden): assert "prompt_cache_key" not in extra assert "cache_control" not in extra - buffers = collect_buffers(iface) - key = f"{GOLDEN_TASK_ID}:{GOLDEN_CALL_TYPE}" - assert len(buffers["openai_compat"][key]) == 6 # 3 turns x (u, a) + history = collect_buffers(iface)["session_histories"][GOLDEN_SESSION_KEY] + assert len(history) == 6 # 3 turns x (u, a) snapshot_scenario("groq", iface, rec) @@ -387,17 +419,28 @@ def test_gemini(golden): iface, rec = golden("gemini", "gemini-2.5-pro") run_scenario(iface) - # Sessionless (no call_type) -> single-turn generate_text. - assert rec.calls[0]["method"] == "generate_text" + posts = [c for c in rec.calls if c["method"] == "generateContent"] + assert len(posts) == 4 + assert all( + p["payload"]["path"] == "models/gemini-2.5-pro:generateContent" for p in posts + ) + + # Sessionless (no call_type) -> single-turn request. + assert len(posts[0]["payload"]["body"]["contents"]) == 1 # Session turns -> multiturn contents array growing by 2 per turn. - multiturns = [c for c in rec.calls if c["method"] == "generate_text_multiturn"] - assert len(multiturns) == 3 - for n, call in enumerate(multiturns, start=1): - contents = call["payload"]["contents"] + for n, call in enumerate(posts[1:], start=1): + contents = call["payload"]["body"]["contents"] assert len(contents) == 2 * (n - 1) + 1 assert contents[-1]["role"] == "user" + # Reasoning default (gemini-2.5-pro budget row) on both the single-turn + # and the multi-turn request. + for post in posts: + config = post["payload"]["body"]["generationConfig"] + assert config["thinkingConfig"] == {"thinkingBudget": 24576} + assert config["maxOutputTokens"] >= 24576 + 8192 + snapshot_scenario("gemini", iface, rec) @@ -447,3 +490,27 @@ def test_remote_ollama(golden): assert "system" not in posts[3]["payload"]["json"] snapshot_scenario("remote", iface, rec) + + +# ───────────────────── Models without a reasoning rule ───────────────────── +# +# A model the reasoning table (agent_core/core/models/reasoning.py) does not +# list must keep receiving exactly the payload it received before reasoning +# defaults existed. These snapshots were recorded before that feature landed +# and must never drift: one scenario per wire that the feature touches. + +UNRULED_MODELS = [ + ("openai", "gpt-4o", "unruled_openai"), + ("anthropic", "claude-3-5-haiku-20241022", "unruled_anthropic"), + ("bedrock", "meta.llama3-3-70b-instruct-v1:0", "unruled_bedrock"), + ("gemini", "gemini-2.0-flash", "unruled_gemini"), + ("openrouter", "anthropic/claude-3.5-sonnet", "unruled_openrouter_claude"), + ("grok", "grok-3", "unruled_grok"), +] + + +@pytest.mark.parametrize("provider, model, snapshot", UNRULED_MODELS) +def test_unruled_model_payload_is_unchanged(golden, provider, model, snapshot): + iface, rec = golden(provider, model) + run_scenario(iface) + snapshot_scenario(snapshot, iface, rec) diff --git a/tests/llm/test_reasoning_rules.py b/tests/llm/test_reasoning_rules.py new file mode 100644 index 00000000..8a45f976 --- /dev/null +++ b/tests/llm/test_reasoning_rules.py @@ -0,0 +1,449 @@ +# -*- coding: utf-8 -*- +"""Per-model reasoning defaults (agent_core/core/models/reasoning.py). + +Four layers: + +1. The table: every row is well formed, uses only the levels its wire can + carry, and is filed under a provider whose transport can send it. +2. The level policy: one level below the strongest, never above "high", + and never below the provider's own default (then the strongest level). +3. Wire rendering: each decision becomes exactly its provider's fields. +4. The transports at CraftBot's real settings (max_tokens 8000): a model with + a rule gets its fields, output cap, and temperature handling; a model + without one is untouched (the golden unruled_* snapshots additionally pin + those payloads byte for byte). +""" + +from __future__ import annotations + +import pytest + +from agent_core.core.impl.llm import reasoning_wire +from agent_core.core.impl.llm.interface import LLMContextOverflowError +from agent_core.core.impl.llm.reasoning_wire import TRANSPORT_WIRES +from agent_core.core.llm.google_gemini_client import _thinking_config +from agent_core.core.models.chatgpt_subscription_client import _translate_request +from agent_core.core.models.reasoning import ( + BUDGET_RUNGS, + BUDGET_WIRES, + KNOWN_LEVELS, + LEVEL_CEILING, + REASONING_RULES, + ReasoningDecision, + ReasoningRule, + ReasoningWire, + resolve_reasoning, + target_level, +) +from agent_core.core.models.registry import get_registry + +from .golden.conftest import GOLDEN_SYSTEM_PROMPT, build_interface + +#: The Anthropic SDK refuses non-streaming requests above this max_tokens. +ANTHROPIC_NON_STREAMING_CEILING = 21_333 +#: Anthropic and OpenRouter reject thinking budgets below this. +MIN_THINKING_BUDGET = 1_024 +#: CraftBot's app-level LLMInterface output cap (app/llm/interface.py). +APP_MAX_TOKENS = 8_000 + + +def _provider(surface: str) -> str: + return "openai" if surface == "openai_subscription" else surface + + +def _auth_mode(surface: str) -> str: + return "subscription" if surface == "openai_subscription" else "api_key" + + +ROWS = [ + (surface, model, rule) + for surface, rows in REASONING_RULES.items() + for model, rule in rows.items() +] +ROW_IDS = [f"{surface}/{model}" for surface, model, _ in ROWS] + + +# ─────────────────────────────── 1. the table ─────────────────────────────── + + +@pytest.mark.parametrize("surface, model, rule", ROWS, ids=ROW_IDS) +def test_row_is_well_formed(surface, model, rule): + assert model == model.strip() and model + assert rule.levels, "a rule needs at least one level" + assert len(set(rule.levels)) == len(rule.levels) + assert all(level in KNOWN_LEVELS for level in rule.levels) + assert list(rule.levels) == sorted(rule.levels, key=KNOWN_LEVELS.index) + assert rule.provider_default is None or rule.provider_default in rule.levels + assert rule.source.startswith("https://") + assert rule.output_tokens >= 0 + if rule.wire in BUDGET_WIRES: + assert rule.levels == BUDGET_RUNGS + assert len(rule.budgets) == len(rule.levels) + assert list(rule.budgets) == sorted(set(rule.budgets)) + assert rule.budgets[0] >= MIN_THINKING_BUDGET + else: + assert rule.budgets == () + + +@pytest.mark.parametrize("surface, model, rule", ROWS, ids=ROW_IDS) +def test_row_is_sendable_by_its_providers_transport(surface, model, rule): + profile = get_registry().get(_provider(surface)) + assert profile is not None, f"{surface!r} is not a registered provider" + assert rule.wire in TRANSPORT_WIRES[profile.wire] + + +@pytest.mark.parametrize("surface, model, rule", ROWS, ids=ROW_IDS) +def test_anthropic_caps_stay_below_the_sdk_non_streaming_ceiling(surface, model, rule): + if rule.wire in (ReasoningWire.ANTHROPIC_ADAPTIVE, ReasoningWire.ANTHROPIC_BUDGET): + assert 0 < rule.output_tokens < ANTHROPIC_NON_STREAMING_CEILING + + +def test_duplicate_model_ids_are_refused(): + from agent_core.core.models import reasoning + + rule = REASONING_RULES["openai"]["gpt-5.2"] + with pytest.raises(ValueError, match="listed twice"): + reasoning._surface({"m": rule}, {"m": rule}) + + +# ─────────────────────────────── 2. the policy ────────────────────────────── + + +@pytest.mark.parametrize( + "levels, provider_default, expected", + [ + (("low", "medium", "high"), None, "medium"), + (("low", "medium", "high", "xhigh"), None, "high"), + # xhigh would be one below max; the ceiling keeps it at high. + (("low", "medium", "high", "xhigh", "max"), "medium", "high"), + (("low", "medium", "high", "max"), "high", "high"), + # medium would be below the provider's own high: use the strongest. + (("low", "medium", "high"), "high", "high"), + (("low", "high", "max"), "max", "max"), + (("high", "max"), "max", "max"), + (("low", "high", "max"), "high", "high"), + (("high",), None, "high"), + (BUDGET_RUNGS, None, "high"), + ], +) +def test_target_level_policy(levels, provider_default, expected): + rule = ReasoningRule( + wire=ReasoningWire.EFFORT, + levels=levels, + provider_default=provider_default, + source="https://example.test", + ) + assert target_level(rule) == expected + + +@pytest.mark.parametrize("surface, model, rule", ROWS, ids=ROW_IDS) +def test_every_row_resolves_within_the_policy(surface, model, rule): + decision = resolve_reasoning(_provider(surface), model, _auth_mode(surface)) + assert decision is not None + assert decision.key == f"{surface}/{model}" + chosen = rule.levels.index(decision.level) + if rule.provider_default is not None: + # Never weaker than what the provider applies when it is omitted. + assert chosen >= rule.levels.index(rule.provider_default) + if LEVEL_CEILING in rule.levels and chosen > rule.levels.index(LEVEL_CEILING): + # Above the ceiling only when the provider's own default already is. + assert rule.provider_default is not None + assert rule.levels.index(rule.provider_default) > rule.levels.index( + LEVEL_CEILING + ) + assert chosen == len(rule.levels) - 1 + if decision.budget_tokens is not None: + assert decision.budget_tokens < decision.output_tokens + + +@pytest.mark.parametrize( + "provider, model, auth_mode, expected", + [ + ("openai", "gpt-5.2-2025-12-11", "api_key", "high"), # CraftBot's default + ("openai", "gpt-5.1", "api_key", "medium"), + ("openai", "gpt-5", "api_key", "medium"), + ("openai", "gpt-6-sol", "api_key", "high"), # never xhigh/max + ("openai", "gpt-5.5", "subscription", "high"), + ("openai", "gpt-6.1-sol", "subscription", "high"), + ("anthropic", "claude-sonnet-4-6", "api_key", "high"), + ("anthropic", "claude-opus-5-5", "api_key", "high"), + ("anthropic", "claude-fable-5-1", "api_key", "high"), + ("openrouter", "openai/gpt-5.2", "api_key", "high"), + ("gemini", "gemini-3.5-flash", "api_key", "medium"), + ("gemini", "gemini-3.1-pro-preview", "api_key", "high"), + ("gemini", "gemini-3-flash-preview", "api_key", "high"), + ("grok", "grok-4.3", "api_key", "high"), + ("grok", "grok-4.5", "api_key", "high"), + ("grok", "grok-4.7", "api_key", "high"), + ("deepseek", "deepseek-flash", "api_key", "high"), + ("glm", "glm-5.3", "api_key", "max"), + ("glm", "glm-5.2", "api_key", "max"), + ("moonshot", "kimi-k3", "api_key", "max"), + ("groq", "openai/gpt-oss-120b", "api_key", "medium"), + ("cerebras", "gpt-oss-120b", "api_key", "medium"), + ], +) +def test_resolved_levels(provider, model, auth_mode, expected): + assert resolve_reasoning(provider, model, auth_mode).level == expected + + +@pytest.mark.parametrize( + "provider, model, budget", + [ + ("anthropic", "claude-haiku-4-5-20251001", 12_288), + ("bedrock", "us.anthropic.claude-haiku-4-5-20251001-v1:0", 12_288), + ("openrouter", "anthropic/claude-sonnet-4.5", 12_288), + ("gemini", "gemini-2.5-pro", 24_576), + ("gemini", "gemini-2.5-flash", 18_432), + ], +) +def test_budget_models_resolve_to_the_high_rung(provider, model, budget): + decision = resolve_reasoning(provider, model) + assert decision.level == "high" + assert decision.budget_tokens == budget + + +@pytest.mark.parametrize( + "provider, model, auth_mode", + [ + ("openai", "gpt-4o", "api_key"), + ("openai", "gpt-5.2", "subscription"), # not in the Codex catalogue + ("openai", "gpt-5.5", "api_key_typo"), # unknown auth mode: api surface + ("openai", "GPT-5.2", "api_key"), # ids are exact + ("openai", " gpt-5.2", "api_key"), + ("openai", "gpt-5.2-pro", "api_key"), + ("grok", "grok-3", "api_key"), + ("mistral", "mistral-large-latest", "api_key"), + ("qwen", "qwen-max", "api_key"), + ("remote", "llama3.2:3b", "api_key"), + ("", "gpt-5.2", "api_key"), + ("openai", None, "api_key"), + (None, "gpt-5.2", "api_key"), + ], +) +def test_models_without_a_rule_resolve_to_none(provider, model, auth_mode): + if auth_mode == "api_key_typo": + # A non-subscription auth mode reads the public-API surface. + assert resolve_reasoning(provider, model, auth_mode).key == "openai/gpt-5.5" + return + assert resolve_reasoning(provider, model, auth_mode) is None + + +def test_output_cap_never_lowers_the_callers_cap(): + decision = resolve_reasoning("openai", "gpt-5.2") + assert decision.output_cap(APP_MAX_TOKENS) == 32_000 + assert decision.output_cap(50_000) == 50_000 + + +# ─────────────────────────────── 3. rendering ─────────────────────────────── + + +def _decision(wire, level="high", budget=None): + return ReasoningDecision( + key="test/model", + wire=wire, + level=level, + budget_tokens=budget, + output_tokens=0, + omit_temperature=False, + ) + + +def test_chat_completions_fields(): + assert reasoning_wire.chat_completions_fields(_decision(ReasoningWire.EFFORT)) == ( + {"reasoning_effort": "high"}, + {}, + ) + assert reasoning_wire.chat_completions_fields( + _decision(ReasoningWire.OPENROUTER_EFFORT) + ) == ({}, {"reasoning": {"effort": "high"}}) + assert reasoning_wire.chat_completions_fields( + _decision(ReasoningWire.OPENROUTER_BUDGET, budget=12_288) + ) == ({}, {"reasoning": {"max_tokens": 12_288}}) + + +def test_anthropic_fields(): + assert reasoning_wire.anthropic_fields( + _decision(ReasoningWire.ANTHROPIC_ADAPTIVE) + ) == {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}} + assert reasoning_wire.anthropic_fields( + _decision(ReasoningWire.ANTHROPIC_BUDGET, budget=12_288) + ) == {"thinking": {"type": "enabled", "budget_tokens": 12_288}} + + +def test_gemini_thinking_kwargs(): + assert reasoning_wire.gemini_thinking_kwargs( + _decision(ReasoningWire.GEMINI_LEVEL, level="medium") + ) == {"thinking_level": "medium"} + assert reasoning_wire.gemini_thinking_kwargs( + _decision(ReasoningWire.GEMINI_BUDGET, budget=24_576) + ) == {"thinking_budget": 24_576} + + +@pytest.mark.parametrize( + "render, wire", + [ + (reasoning_wire.chat_completions_fields, ReasoningWire.ANTHROPIC_ADAPTIVE), + (reasoning_wire.anthropic_fields, ReasoningWire.EFFORT), + (reasoning_wire.gemini_thinking_kwargs, ReasoningWire.OPENROUTER_EFFORT), + ], +) +def test_a_wire_the_transport_cannot_send_raises(render, wire): + with pytest.raises(ValueError, match="wrong provider"): + render(_decision(wire)) + + +def test_gemini_thinking_config_builder(): + assert _thinking_config(None, None) is None + assert _thinking_config(512, None) == {"thinkingBudget": 512} + assert _thinking_config(None, "medium") == {"thinkingLevel": "medium"} + with pytest.raises(ValueError, match="not both"): + _thinking_config(512, "medium") + + +def test_codex_translator_uses_the_resolved_effort(): + messages = [{"role": "system", "content": "s"}, {"role": "user", "content": "u"}] + ruled = _translate_request( + {"model": "gpt-5.5", "messages": messages, "reasoning_effort": "high"}, "k" + ) + assert ruled["reasoning"] == {"effort": "high", "summary": "auto"} + # No rule: the backend still requires the block, so today's medium stays. + unruled = _translate_request({"model": "gpt-5.4", "messages": messages}, "k") + assert unruled["reasoning"] == {"effort": "medium", "summary": "auto"} + + +# ────────────────────── 4. transports at CraftBot's settings ───────────────── + + +def _one_call(monkeypatch, provider, model, auth_mode="api_key"): + iface, rec = build_interface(monkeypatch, provider, model) + iface.max_tokens = APP_MAX_TOKENS + iface._auth_mode = auth_mode + iface.generate_response(system_prompt=GOLDEN_SYSTEM_PROMPT, user_prompt="hi") + assert len(rec.calls) == 1 + return iface, rec.calls[0]["payload"] + + +def test_openai_ruled_model(monkeypatch): + _, payload = _one_call(monkeypatch, "openai", "gpt-5.2-2025-12-11") + assert payload["reasoning_effort"] == "high" + assert payload["max_completion_tokens"] == 32_000 + assert "temperature" not in payload + assert payload["response_format"] == {"type": "json_object"} + + +def test_openai_unruled_model(monkeypatch): + _, payload = _one_call(monkeypatch, "openai", "gpt-4o") + assert "reasoning_effort" not in payload + assert payload["max_completion_tokens"] == APP_MAX_TOKENS + + +def test_subscription_request_carries_the_codex_row(monkeypatch): + _, payload = _one_call(monkeypatch, "openai", "gpt-5.5", auth_mode="subscription") + assert payload["reasoning_effort"] == "high" + # Codex rows leave the (dropped) output cap alone. + assert payload["max_completion_tokens"] == APP_MAX_TOKENS + assert _translate_request(payload, "k")["reasoning"]["effort"] == "high" + + +def test_groq_cap_is_clamped_by_the_profile_and_temperature_kept(monkeypatch): + _, payload = _one_call(monkeypatch, "groq", "openai/gpt-oss-120b") + assert payload["reasoning_effort"] == "medium" + assert payload["max_completion_tokens"] == 32_000 # <= Groq's 32,768 clamp + assert payload["temperature"] == 0.0 + assert payload["response_format"] == {"type": "json_object"} + + +def test_cerebras_cap_is_clamped_by_the_profile(monkeypatch): + _, payload = _one_call(monkeypatch, "cerebras", "gpt-oss-120b") + assert payload["reasoning_effort"] == "medium" + assert payload["max_completion_tokens"] == 32_000 + + +def test_glm_bad_model_is_pinned_at_its_max(monkeypatch): + _, payload = _one_call(monkeypatch, "glm", "glm-5.3") + assert payload["reasoning_effort"] == "max" + assert payload["max_tokens"] == 32_000 + + +def test_openrouter_openai_row(monkeypatch): + _, payload = _one_call(monkeypatch, "openrouter", "openai/gpt-5.2") + assert payload["extra_body"]["reasoning"] == {"effort": "high"} + assert "reasoning_effort" not in payload + assert "temperature" not in payload + assert payload["max_tokens"] == 32_000 + + +def test_anthropic_ruled_model(monkeypatch): + _, payload = _one_call(monkeypatch, "anthropic", "claude-opus-5-5") + assert payload["thinking"] == {"type": "adaptive"} + assert payload["output_config"] == {"effort": "high"} + assert payload["max_tokens"] == 21_000 + assert "extra_body" not in payload # no temperature + + +def test_anthropic_unruled_model(monkeypatch): + _, payload = _one_call(monkeypatch, "anthropic", "claude-3-5-haiku-20241022") + assert "thinking" not in payload and "output_config" not in payload + assert payload["max_tokens"] == 16_384 + assert payload["extra_body"] == {"temperature": 0.0} + + +def test_bedrock_adaptive_row(monkeypatch): + _, payload = _one_call(monkeypatch, "bedrock", "us.anthropic.claude-opus-4-8") + assert payload["additionalModelRequestFields"] == { + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "high"}, + } + assert payload["inferenceConfig"] == {"maxTokens": 21_000} + + +def test_bedrock_unruled_model(monkeypatch): + _, payload = _one_call(monkeypatch, "bedrock", "meta.llama3-3-70b-instruct-v1:0") + assert "additionalModelRequestFields" not in payload + assert payload["inferenceConfig"] == { + "temperature": 0.0, + "maxTokens": APP_MAX_TOKENS, + } + + +def test_gemini_level_row(monkeypatch): + _, payload = _one_call(monkeypatch, "gemini", "gemini-3.5-flash") + config = payload["body"]["generationConfig"] + assert config["thinkingConfig"] == {"thinkingLevel": "medium"} + assert config["maxOutputTokens"] == 32_768 + + +def test_gemini_caller_budget_wins_over_the_default(monkeypatch): + iface, rec = build_interface(monkeypatch, "gemini", "gemini-3.5-flash") + iface.max_tokens = APP_MAX_TOKENS + iface._begin_call(thinking_budget=512) + iface._generate_response_sync(GOLDEN_SYSTEM_PROMPT, "hi") + config = rec.calls[0]["payload"]["body"]["generationConfig"] + assert config["thinkingConfig"] == {"thinkingBudget": 512} + assert config["maxOutputTokens"] == APP_MAX_TOKENS + + +def test_context_check_reserves_the_reasoning_cap(monkeypatch): + import app.config as app_config + + monkeypatch.setattr(app_config, "get_context_window", lambda: 40_000) + prompt = "word " * 12_000 # ~12k tokens: fits 40k - 8k, not 40k - 32k + + unruled, _ = build_interface(monkeypatch, "openai", "gpt-4o") + unruled.max_tokens = APP_MAX_TOKENS + assert unruled._output_reservation() == APP_MAX_TOKENS + unruled._check_context_fits(None, prompt) + + ruled, _ = build_interface(monkeypatch, "openai", "gpt-5.2-2025-12-11") + ruled.max_tokens = APP_MAX_TOKENS + assert ruled._output_reservation() == 32_000 + with pytest.raises(LLMContextOverflowError, match="32000 reserved for output"): + ruled._check_context_fits(None, prompt) + + +def test_decision_follows_a_model_switch(monkeypatch): + iface, _ = build_interface(monkeypatch, "openai", "gpt-4o") + assert iface.reasoning_decision() is None + iface.model = "gpt-5.2" + assert iface.reasoning_decision().level == "high" diff --git a/tests/llm/test_reliability.py b/tests/llm/test_reliability.py new file mode 100644 index 00000000..36006e56 --- /dev/null +++ b/tests/llm/test_reliability.py @@ -0,0 +1,244 @@ +# -*- coding: utf-8 -*- +"""Phase 5 reliability tests: credential pools (FR-7) and cross-provider +fallback (FR-9). + +Fallback assertions include the NFR-3 property from the spec's Phase 5 +acceptance: after a fallback turn, the PRIMARY interface's session buffers +are untouched and the fallback interface accumulates its own, so returning +to the primary next turn stays cache-warm. +""" + +from __future__ import annotations + +import pytest + +from agent_core.core.impl.llm.errors import LLMConsecutiveFailureError + +from .golden.conftest import ( + GOLDEN_SYSTEM_PROMPT, + build_interface, +) + + +# ─────────────────────── credential pool state machine ─────────────────────── + + +@pytest.fixture() +def pool(monkeypatch, tmp_path): + from agent_core.core.models import credentials + + monkeypatch.setattr( + credentials, "_state_path", lambda: tmp_path / "pool_state.json" + ) + credentials.reset_state_for_tests() + + keys = {"primary": "key-A", "extras": ["key-B", "key-C"]} + import app.config as app_config + + monkeypatch.setattr(app_config, "get_api_key", lambda p: keys["primary"]) + monkeypatch.setattr( + app_config, "get_extra_api_keys", lambda p: list(keys["extras"]) + ) + return credentials + + +def test_fill_first_serves_primary(pool): + assert pool.resolve("anthropic") == "key-A" + assert pool.has_pool("anthropic") is True + + +def test_rate_limit_cools_after_second_consecutive(pool): + pool.resolve("anthropic") + pool.note_failure("anthropic", "rate_limit") # 1st: no cooldown yet + assert pool.resolve("anthropic") == "key-A" + pool.note_failure("anthropic", "rate_limit") # 2nd: cool 60s + assert pool.resolve("anthropic") == "key-B" + + +def test_billing_rotates_immediately(pool): + pool.resolve("anthropic") + pool.note_failure("anthropic", "credit") + assert pool.resolve("anthropic") == "key-B" + pool.note_failure("anthropic", "credit") # B billing-cooled too + assert pool.resolve("anthropic") == "key-C" + + +def test_auth_rotates_and_success_clears(pool): + pool.resolve("anthropic") + pool.note_failure("anthropic", "auth") + assert pool.resolve("anthropic") == "key-B" + pool.note_success("anthropic") # clears B's (empty) record, keeps A cooling + assert pool.resolve("anthropic") == "key-B" + + +def test_all_cooling_fails_open_to_primary(pool): + for _ in range(3): + pool.resolve("anthropic") + pool.note_failure("anthropic", "credit") + assert pool.resolve("anthropic") == "key-A" + + +def test_transient_categories_do_not_cool(pool): + pool.resolve("anthropic") + pool.note_failure("anthropic", "server") + pool.note_failure("anthropic", "connection") + pool.note_failure("anthropic", "unknown") + assert pool.resolve("anthropic") == "key-A" + + +def test_single_key_provider_is_untouched(pool, monkeypatch): + import app.config as app_config + + monkeypatch.setattr(app_config, "get_extra_api_keys", lambda p: []) + assert pool.has_pool("anthropic") is False + pool.resolve("anthropic") + pool.note_failure("anthropic", "credit") + assert pool.resolve("anthropic") == "key-A" + + +# ─────────────────────── cross-provider fallback ─────────────────────── + + +def _break_client(iface): + """Make the fake OpenAI-compat client raise on every call.""" + + def failing_create(**kwargs): + raise RuntimeError("primary provider down") + + iface.client.chat = type(iface.client.chat)( + completions=type(iface.client.chat.completions)(create=failing_create) + ) + + +@pytest.fixture() +def fallback_pair(monkeypatch): + """Primary openai interface (broken) + anthropic fallback (healthy), + with the chain configured and the fallback interface injected.""" + primary, primary_rec = build_interface(monkeypatch, "openai", "gpt-5.2-2025-12-11") + fallback, fallback_rec = build_interface( + monkeypatch, "anthropic", "claude-sonnet-4-6" + ) + _break_client(primary) + + import app.config as app_config + + monkeypatch.setattr( + app_config, "get_fallback_providers", lambda: ["anthropic"] + ) + monkeypatch.setattr(app_config, "get_api_key", lambda p: "k") + # Inject the pre-built fallback interface (its factory patch would + # otherwise have been torn down by build_interface's second call). + primary._fallback_interfaces["anthropic"] = fallback + return primary, primary_rec, fallback, fallback_rec + + +def test_fallback_serves_sessionless_turn(fallback_pair): + primary, _prec, fallback, frec = fallback_pair + + out = primary.generate_response( + system_prompt=GOLDEN_SYSTEM_PROMPT, user_prompt="hello" + ) + assert out # turn served by the fallback + assert any(c["method"] == "messages.create" for c in frec.calls) + # Served turn == success: the primary's counter must not tick. + assert primary.consecutive_failures == 0 + + +def test_fallback_session_turn_preserves_primary_buffers(fallback_pair): + primary, _prec, fallback, frec = fallback_pair + + # Prime one successful-looking primary session state by hand: the + # buffers below must survive the fallback turn untouched. + primary.create_session_cache("t1", "reasoning", GOLDEN_SYSTEM_PROMPT) + primary._session_histories["t1:reasoning"] = [ + {"role": "user", "content": "u1"}, + {"role": "assistant", "content": "a1"}, + ] + before = [dict(m) for m in primary._session_histories["t1:reasoning"]] + + out = primary.generate_response_with_session("t1", "reasoning", "turn 2") + assert out # served by anthropic fallback + + # NFR-3: primary buffers untouched; fallback accumulated its own. + assert primary._session_histories["t1:reasoning"] == before + assert fallback._session_histories["t1:reasoning"] + assert primary.consecutive_failures == 0 + + +def test_fallback_off_preserves_error_contract(monkeypatch): + """No chain configured -> exact historical failure behavior (NFR-1).""" + primary, rec = build_interface(monkeypatch, "openai", "gpt-5.2-2025-12-11") + _break_client(primary) + import app.config as app_config + + monkeypatch.setattr(app_config, "get_fallback_providers", lambda: []) + + for _ in range(4): + with pytest.raises(Exception): + primary.generate_response(system_prompt="s", user_prompt="u") + assert primary.consecutive_failures == 4 + with pytest.raises(LLMConsecutiveFailureError): + primary.generate_response(system_prompt="s", user_prompt="u") + + +def test_strict_turn_after_reinitialize_skips_fallback(fallback_pair): + primary, _prec, fallback, frec = fallback_pair + + # Simulate the flag reinitialize() sets on explicit selection. + primary._suppress_fallback_once = True + with pytest.raises(Exception): + primary.generate_response(system_prompt="s", user_prompt="u") + assert not any(c["method"] == "messages.create" for c in frec.calls) + + # Next turn: fallback active again. + out = primary.generate_response(system_prompt="s", user_prompt="u") + assert out + assert any(c["method"] == "messages.create" for c in frec.calls) + + +def test_total_outage_terminates_without_recursion(monkeypatch): + """Multi-provider chain with EVERYTHING down must terminate in one chain + walk (audit finding 2026-08-17: fallback instances used to walk their + own chains, nesting primary -> fb -> fb-of-fb without bound).""" + primary, _ = build_interface(monkeypatch, "openai", "gpt-5.2-2025-12-11") + fb1, _ = build_interface(monkeypatch, "anthropic", "claude-sonnet-4-6") + fb2, _ = build_interface(monkeypatch, "deepseek", "deepseek-chat") + _break_client(primary) + fb1.messages_broken = True + + def failing_messages(**kwargs): + raise RuntimeError("anthropic down") + + fb1._anthropic_client.messages = type(fb1._anthropic_client.messages)( + create=failing_messages + ) + _break_client(fb2) + + import app.config as app_config + + monkeypatch.setattr( + app_config, "get_fallback_providers", lambda: ["anthropic", "deepseek"] + ) + monkeypatch.setattr(app_config, "get_api_key", lambda p: "k") + primary._fallback_interfaces = {"anthropic": fb1, "deepseek": fb2} + fb1._is_fallback_instance = True + fb2._is_fallback_instance = True + + with pytest.raises(Exception) as exc_info: + primary.generate_response(system_prompt="s", user_prompt="u") + # Terminated with a plain failure, not a RecursionError. + assert not isinstance(exc_info.value, RecursionError) + assert primary.consecutive_failures == 1 + + # Structural guarantee: fallback instances never expose a chain. + assert fb1._fallback_chain() == [] + assert fb2._fallback_chain() == [] + + +def test_fallback_notice_hook_fires(fallback_pair): + primary, _prec, fallback, _frec = fallback_pair + events = [] + primary._on_fallback = lambda frm, to, reason: events.append((frm, to, reason)) + + primary.generate_response(system_prompt="s", user_prompt="u") + assert events == [("openai", "anthropic", events[0][2])] diff --git a/tests/test_bedrock_token_normalization.py b/tests/test_bedrock_token_normalization.py index c0860885..58ce2a9d 100644 --- a/tests/test_bedrock_token_normalization.py +++ b/tests/test_bedrock_token_normalization.py @@ -52,6 +52,9 @@ def test_llm_bedrock_input_includes_cache_and_cached_is_reads_only(): iface = LLMInterface.__new__(LLMInterface) reported = _stub_common(iface) iface._call_log_to_db = lambda *a, **kw: None + # Read by the per-model reasoning lookup every transport performs. + iface._auth_mode = "api_key" + iface._reasoning_logged_for = None result = iface._generate_bedrock(None, "hi") From de5902c3f51d2d70552c7f8462cc4cff00bb267c Mon Sep 17 00:00:00 2001 From: CraftBot Date: Fri, 2 Oct 2026 09:10:18 +0900 Subject: [PATCH 2/5] create effort setting knob --- agent_core/core/impl/action/context.py | 11 +- agent_core/core/impl/action/executor.py | 15 +- agent_core/core/impl/llm/interface.py | 72 +- agent_core/core/impl/llm/reasoning_wire.py | 86 +-- .../impl/llm/transports/bedrock_converse.py | 6 +- agent_core/core/impl/session/manager.py | 30 + agent_core/core/models/reasoning.py | 637 +++++++++++++----- agent_core/core/session/session.py | 17 + agent_core/core/state/session.py | 33 +- app/agent_base.py | 70 +- app/triggers/runtime.py | 6 +- app/ui_layer/adapters/browser_adapter.py | 83 ++- .../src/components/Chat/Chat.module.css | 6 + .../frontend/src/components/Chat/Chat.tsx | 78 ++- .../Chat/ReasoningPicker.module.css | 102 +++ .../src/components/Chat/ReasoningPicker.tsx | 159 +++++ .../frontend/src/components/layout/NavBar.tsx | 1 + .../src/contexts/WebSocketContext.tsx | 59 +- .../browser/frontend/src/locales/en/chat.json | 20 + .../browser/frontend/src/locales/es/chat.json | 20 + .../browser/frontend/src/locales/id/chat.json | 20 + .../browser/frontend/src/locales/ja/chat.json | 20 + .../browser/frontend/src/locales/ko/chat.json | 20 + .../frontend/src/locales/zh-CN/chat.json | 20 + .../frontend/src/locales/zh-TW/chat.json | 20 + .../browser/frontend/src/store/index.ts | 2 + .../frontend/src/store/resources/catalog.ts | 7 + .../frontend/src/store/selectors/reasoning.ts | 3 + .../src/store/slices/reasoningSlice.ts | 52 ++ .../frontend/src/store/uiState/catalog.ts | 9 +- .../browser/frontend/src/types/index.ts | 19 + app/ui_layer/controller/ui_controller.py | 14 +- scripts/probe_reasoning.py | 60 +- tests/llm/test_reasoning_rules.py | 632 ++++++++++++++--- 34 files changed, 2003 insertions(+), 406 deletions(-) create mode 100644 app/ui_layer/browser/frontend/src/components/Chat/ReasoningPicker.module.css create mode 100644 app/ui_layer/browser/frontend/src/components/Chat/ReasoningPicker.tsx create mode 100644 app/ui_layer/browser/frontend/src/store/selectors/reasoning.ts create mode 100644 app/ui_layer/browser/frontend/src/store/slices/reasoningSlice.ts diff --git a/agent_core/core/impl/action/context.py b/agent_core/core/impl/action/context.py index 66f0cde0..96039512 100644 --- a/agent_core/core/impl/action/context.py +++ b/agent_core/core/impl/action/context.py @@ -9,9 +9,9 @@ Scope rules: - Set only by the internal executors (``_atomic_action_internal*``), reset in a ``finally`` — never leaks across actions. - - Sync actions run in a thread pool where the caller's context does NOT - propagate, so the executor wraps the call and sets the var inside the - worker thread (see ``run_with_input_context``). + - Sync actions run in a thread pool inside a copy of the caller's + context; the executor wraps the call and sets the var inside the worker + thread (see ``run_with_input_context``). - Sandboxed (subprocess) actions cannot see it at all — helpers must treat a ``None`` value as "no context available". """ @@ -31,8 +31,9 @@ def run_with_input_context( ) -> dict: """Call a sync action with ``current_input_data`` set for its duration. - Used as the thread-pool target: the worker thread has its own context, - so the var must be set (and reset) inside the thread, not the caller. + Used as the thread-pool target, run inside a copy of the caller's + context: the var is set (and reset) inside the thread so it scopes to + this action alone. """ token = current_input_data.set(input_data) try: diff --git a/agent_core/core/impl/action/executor.py b/agent_core/core/impl/action/executor.py index 5b735dfd..6d15b734 100644 --- a/agent_core/core/impl/action/executor.py +++ b/agent_core/core/impl/action/executor.py @@ -14,6 +14,7 @@ """ import asyncio +import contextvars import importlib import json import os @@ -634,14 +635,20 @@ async def _atomic_action_internal_async( finally: current_input_data.reset(ctx_token) else: - # Sync function - run in thread pool to avoid blocking. The - # worker thread doesn't inherit this context, so the wrapper - # sets current_input_data inside the thread. + # Sync function - run in thread pool to avoid blocking. A pool + # thread does not inherit the caller's context, so the action + # runs inside a copy of it (as asyncio.to_thread does): the bound + # session, log context, and other context-scoped state follow the + # action into the thread, and the wrapper sets current_input_data + # there. logger.debug( f"[SYNC] Action '{action_name}' is sync, running in thread pool" ) thread_future = THREAD_POOL.submit( - run_with_input_context, function_to_call, input_data + contextvars.copy_context().run, + run_with_input_context, + function_to_call, + input_data, ) try: execution_result = await asyncio.wrap_future(thread_future) diff --git a/agent_core/core/impl/llm/interface.py b/agent_core/core/impl/llm/interface.py index 7a97fd51..520330cf 100644 --- a/agent_core/core/impl/llm/interface.py +++ b/agent_core/core/impl/llm/interface.py @@ -42,11 +42,19 @@ RecordLLMCallHook, ) from agent_core.core.impl.llm import transports as _transports -from agent_core.core.models.reasoning import ReasoningDecision, resolve_reasoning +from agent_core.core.models.reasoning import ( + ReasoningChoice, + ReasoningDecision, + ReasoningOptions, + default_choice, + reasoning_options as _reasoning_options, + resolve_reasoning, +) from agent_core.core.models.registry import ( get_registry as _get_registry, session_cc_providers as _session_cc_providers, ) +from agent_core.core.state.session import StateSession # Logging setup - use shared agent_core logger for consistency from agent_core.utils.logger import logger @@ -588,6 +596,7 @@ def _begin_call( call_type: Optional[str] = None, task_id: Optional[str] = None, thinking_budget: Optional[int] = None, + reasoning_choice: Optional[ReasoningChoice] = None, ) -> None: """Stamp per-call identity + start time into the context for capture. @@ -598,6 +607,9 @@ def _begin_call( ``thinking_budget`` (when set) is read by the Gemini transport to cap reasoning tokens; other transports never look at it. + + ``reasoning_choice`` (when set) overrides the bound session's choice + for this call; see ``reasoning_decision``. """ _llm_call_ctx.set( { @@ -605,6 +617,7 @@ def _begin_call( "call_type": call_type, "task_id": task_id, "thinking_budget": thinking_budget, + "reasoning_choice": reasoning_choice, "start": time.perf_counter(), } ) @@ -804,16 +817,52 @@ def _check_context_fits( f"{self.provider}/{self.model}." ) + def reasoning_options(self) -> Optional[ReasoningOptions]: + """What the chat-input reasoning picker offers for the model in use. + + None when the model has no rule (its reasoning is not adjustable). + """ + return _reasoning_options(self.provider, self.model, self._auth_mode) + + def default_reasoning_choice(self) -> ReasoningChoice: + """The choice a new session starts at with this interface's model.""" + return default_choice(self.provider, self.model, self._auth_mode) + + def reasoning_choice(self) -> ReasoningChoice: + """The reasoning choice that applies to the call in flight. + + A choice passed for this call wins (work for a chat that has no + session yet: the draft view); otherwise the choice of the session + the context is bound to (StateSession.bind, set around each + session's loop); otherwise, for work that belongs to no session, + the model's default level. + """ + explicit = _llm_call_ctx.get().get("reasoning_choice") + if explicit is not None: + return explicit + state = StateSession.bound() + if ( + state is not None + and state.current_session is not None + and state.current_session.reasoning_effort is not None + ): + return ReasoningChoice(state.current_session.reasoning_effort) + return self.default_reasoning_choice() + def reasoning_decision(self) -> Optional[ReasoningDecision]: - """Reasoning default for the current provider and model. + """What the call in flight sends for reasoning. None means the model has no rule in agent_core/core/models/reasoning.py and its requests carry no reasoning parameter. Resolved on every call - so a model switch (reinitialize) or a fallback interface uses its own - row; logged once per distinct model. + so a model switch (reinitialize), a fallback interface, or a changed + session choice applies from the next request; logged once per + distinct model and choice. """ - decision = resolve_reasoning(self.provider, self.model, self._auth_mode) - log_key = (self.provider, self.model, self._auth_mode) + choice = self.reasoning_choice() + decision = resolve_reasoning( + self.provider, self.model, self._auth_mode, choice + ) + log_key = (self.provider, self.model, self._auth_mode, choice) if log_key != self._reasoning_logged_for: self._reasoning_logged_for = log_key if decision is None: @@ -1008,6 +1057,7 @@ async def generate_response_async( prompt_name: Optional[str] = None, json_mode: bool = True, thinking_budget: Optional[int] = None, + reasoning_choice: Optional[ReasoningChoice] = None, ) -> str: """Async wrapper that defers the blocking call to a worker thread. @@ -1017,10 +1067,18 @@ async def generate_response_async( ``thinking_budget`` caps reasoning tokens on providers that expose a thinking budget (Gemini). It rides the per-call context and is a no-op for every other provider; leave it None (the default) for normal calls. + + ``reasoning_choice`` overrides the bound session's reasoning choice + for this call; pass it only for work done for a chat that has no + session yet. """ # Stamp the context here, in the caller's context, so asyncio.to_thread # copies it into the worker thread where the capture runs. - self._begin_call(prompt_name=prompt_name, thinking_budget=thinking_budget) + self._begin_call( + prompt_name=prompt_name, + thinking_budget=thinking_budget, + reasoning_choice=reasoning_choice, + ) return await asyncio.to_thread( self._generate_response_sync, system_prompt, diff --git a/agent_core/core/impl/llm/reasoning_wire.py b/agent_core/core/impl/llm/reasoning_wire.py index 9203801b..32d9189e 100644 --- a/agent_core/core/impl/llm/reasoning_wire.py +++ b/agent_core/core/impl/llm/reasoning_wire.py @@ -1,49 +1,36 @@ # -*- coding: utf-8 -*- -"""Render a reasoning default into request fields, one function per transport. +"""Render a reasoning decision into request fields, one function per transport. -The table and the level policy live in agent_core/core/models/reasoning.py; -this module only knows how each transport spells a ``ReasoningDecision``. -Callers skip these functions entirely when a model has no rule, so a model -without a rule never gains a field here. +The table, the choices, and the level policy live in +agent_core/core/models/reasoning.py; this module only knows how each +transport spells a ``ReasoningDecision``. Callers skip these functions +entirely when a model has no rule, so a model without a rule never gains a +field here. A decision that sends nothing (the provider default, or off on a +model whose default is no reasoning) renders as no fields. A decision whose wire the transport cannot express is a bug in the rules -table (a row filed under the wrong provider), so it raises instead of being -dropped silently. tests/llm/test_reasoning_rules.py checks every row against -``TRANSPORT_WIRES`` so this never reaches production. +table (a row filed under the wrong provider, or an explicit off on a wire +without a disable form), so it raises instead of being dropped silently. +tests/llm/test_reasoning_rules.py renders every choice on every row through +its provider's function, so this never reaches production. """ from __future__ import annotations -from typing import Any, Dict, FrozenSet, Mapping, Tuple +from typing import Any, Dict, Tuple from agent_core.core.models.reasoning import ReasoningDecision, ReasoningWire -#: Reasoning wires each transport (ProviderProfile.wire) can express. -TRANSPORT_WIRES: Mapping[str, FrozenSet[ReasoningWire]] = { - "chat_completions": frozenset( - { - ReasoningWire.EFFORT, - ReasoningWire.OPENROUTER_EFFORT, - ReasoningWire.OPENROUTER_BUDGET, - } - ), - "anthropic_messages": frozenset( - {ReasoningWire.ANTHROPIC_ADAPTIVE, ReasoningWire.ANTHROPIC_BUDGET} - ), - "bedrock_converse": frozenset( - {ReasoningWire.ANTHROPIC_ADAPTIVE, ReasoningWire.ANTHROPIC_BUDGET} - ), - "gemini_native": frozenset( - {ReasoningWire.GEMINI_LEVEL, ReasoningWire.GEMINI_BUDGET} - ), -} +#: The disable value of the OpenAI-style effort parameter. +EFFORT_OFF = "none" def _unsupported(decision: ReasoningDecision, transport: str) -> ValueError: + what = "an explicit off" if decision.off else f"wire {decision.wire.value!r}" return ValueError( - f"Reasoning rule {decision.key!r} uses wire {decision.wire.value!r}, " - f"which the {transport} transport cannot send. The row is filed under " - f"the wrong provider in agent_core/core/models/reasoning.py." + f"Reasoning rule {decision.key!r} asks for {what}, which the " + f"{transport} transport cannot send. Fix the row in " + f"agent_core/core/models/reasoning.py." ) @@ -52,10 +39,20 @@ def chat_completions_fields( ) -> Tuple[Dict[str, Any], Dict[str, Any]]: """Fields for a Chat Completions request: (top-level kwargs, extra_body).""" if decision.wire is ReasoningWire.EFFORT: + if decision.off: + return {"reasoning_effort": EFFORT_OFF}, {} + if decision.level is None: + return {}, {} return {"reasoning_effort": decision.level}, {} if decision.wire is ReasoningWire.OPENROUTER_EFFORT: + if decision.off: + return {}, {"reasoning": {"effort": EFFORT_OFF}} + if decision.level is None: + return {}, {} return {}, {"reasoning": {"effort": decision.level}} - if decision.wire is ReasoningWire.OPENROUTER_BUDGET: + if decision.wire is ReasoningWire.OPENROUTER_BUDGET and not decision.off: + if decision.level is None: + return {}, {} return {}, {"reasoning": {"max_tokens": decision.budget_tokens}} raise _unsupported(decision, "chat_completions") @@ -66,22 +63,33 @@ def anthropic_fields(decision: ReasoningDecision) -> Dict[str, Any]: The same dict is the ``additionalModelRequestFields`` of a Bedrock Converse request for Claude, which forwards these keys to the model. """ + if decision.wire not in ( + ReasoningWire.ANTHROPIC_ADAPTIVE, + ReasoningWire.ANTHROPIC_BUDGET, + ): + raise _unsupported(decision, "anthropic_messages/bedrock_converse") + if decision.off: + return {"thinking": {"type": "disabled"}} + if decision.level is None: + return {} if decision.wire is ReasoningWire.ANTHROPIC_ADAPTIVE: return { "thinking": {"type": "adaptive"}, "output_config": {"effort": decision.level}, } - if decision.wire is ReasoningWire.ANTHROPIC_BUDGET: - return { - "thinking": {"type": "enabled", "budget_tokens": decision.budget_tokens} - } - raise _unsupported(decision, "anthropic_messages/bedrock_converse") + return {"thinking": {"type": "enabled", "budget_tokens": decision.budget_tokens}} def gemini_thinking_kwargs(decision: ReasoningDecision) -> Dict[str, Any]: """Keyword arguments for the GeminiClient text-generation methods.""" - if decision.wire is ReasoningWire.GEMINI_LEVEL: - return {"thinking_level": decision.level} if decision.wire is ReasoningWire.GEMINI_BUDGET: + if decision.off: + return {"thinking_budget": 0} + if decision.level is None: + return {} return {"thinking_budget": decision.budget_tokens} + if decision.wire is ReasoningWire.GEMINI_LEVEL and not decision.off: + if decision.level is None: + return {} + return {"thinking_level": decision.level} raise _unsupported(decision, "gemini_native") diff --git a/agent_core/core/impl/llm/transports/bedrock_converse.py b/agent_core/core/impl/llm/transports/bedrock_converse.py index bbf3abe2..7c41d07d 100644 --- a/agent_core/core/impl/llm/transports/bedrock_converse.py +++ b/agent_core/core/impl/llm/transports/bedrock_converse.py @@ -88,9 +88,9 @@ def generate( # through additionalModelRequestFields. Thinking tokens count # against maxTokens, and thinking is incompatible with (4.5/4.6) # or rejects (4.7+) a non-default temperature. - converse_kwargs["additionalModelRequestFields"] = ( - reasoning_wire.anthropic_fields(reasoning) - ) + reasoning_fields = reasoning_wire.anthropic_fields(reasoning) + if reasoning_fields: + converse_kwargs["additionalModelRequestFields"] = reasoning_fields converse_kwargs["inferenceConfig"]["maxTokens"] = reasoning.output_cap( iface.max_tokens ) diff --git a/agent_core/core/impl/session/manager.py b/agent_core/core/impl/session/manager.py index 35c12927..bd18bbdd 100644 --- a/agent_core/core/impl/session/manager.py +++ b/agent_core/core/impl/session/manager.py @@ -33,6 +33,7 @@ ) from agent_core.core.state import StateSession from agent_core.core.impl.llm import LLMCallType +from agent_core.core.models.reasoning import ReasoningChoice, default_choice from agent_core.utils.logger import logger @@ -137,6 +138,7 @@ def create_session( selected_skills: Optional[List[str]] = None, agent_app_project_id: Optional[str] = None, gui_mode: bool = False, + reasoning_effort: Optional[ReasoningChoice] = None, ) -> Session: """ Create a new persistent session. @@ -149,6 +151,9 @@ def create_session( selected_skills: Skills to preload (slash-command entry, Agent App). agent_app_project_id: Backing project for agent_app sessions. gui_mode: Whether the session starts in GUI mode. + reasoning_effort: The session's reasoning choice (the draft + chat's picker value when a draft becomes a session); None + starts it at the default level of the model in use. Returns: The created Session. @@ -179,6 +184,7 @@ def create_session( workspace_dir=str(workspace_dir), agent_app_project_id=agent_app_project_id, gui_mode=gui_mode, + reasoning_effort=(reasoning_effort or self._default_reasoning()).value, ) self.sessions[sid] = session @@ -275,6 +281,18 @@ def rename_session(self, session_id: str, title: str) -> bool: self._persist(session) return True + def set_reasoning_effort(self, session_id: str, choice: ReasoningChoice) -> bool: + """Set a session's reasoning choice (the chat input's picker). + + Takes effect from the session's next LLM request. + """ + session = self.sessions.get(session_id) + if not session: + return False + session.reasoning_effort = choice.value + self._persist(session) + return True + # ─────────────────────── Restore ───────────────────────────────────────── def restore_session(self, session: Session) -> Session: @@ -299,6 +317,11 @@ def restore_session(self, session: Session) -> Session: StateSession.start( session.id, current_session=session, gui_mode=session.gui_mode ) + if session.reasoning_effort is None: + # Saved without a valid reasoning choice: start it where a new + # session would. + session.reasoning_effort = self._default_reasoning().value + self._persist(session) return session # ─────────────────────── Todo Management ───────────────────────────────── @@ -542,6 +565,13 @@ def _create_session_caches(self, session_id: str) -> None: # ─────────────────────── Internal Helpers ──────────────────────────────── + def _default_reasoning(self) -> ReasoningChoice: + """The reasoning choice a session starts at: the default level of the + model in use.""" + if self.llm_interface is None: + return default_choice(None, None) + return self.llm_interface.default_reasoning_choice() + def _persist(self, session: Session) -> None: """Persist session state via hook.""" if self._on_session_persist: diff --git a/agent_core/core/models/reasoning.py b/agent_core/core/models/reasoning.py index 3f2301ce..8d6d5baf 100644 --- a/agent_core/core/models/reasoning.py +++ b/agent_core/core/models/reasoning.py @@ -1,5 +1,5 @@ # -*- coding: utf-8 -*- -"""Default reasoning effort ("thinking") per model. +"""Reasoning effort ("thinking") per model, chosen per chat session. Why this exists --------------- @@ -11,18 +11,34 @@ models of the SAME provider, and a value a model does not accept is a hard HTTP 400 on the first request. -So defaults live in a hard-coded table keyed by provider and EXACT model id -(no prefix or substring matching). A model that is not in the table gets no -reasoning parameter at all, and its request stays byte-identical to the one -sent before this module existed (pinned by the ``unruled_*`` golden payload -snapshots under tests/llm/golden/). - -Choosing the level ------------------- +So the capabilities live in a hard-coded table keyed by provider and EXACT +model id (no prefix or substring matching). A model that is not in the table +gets no reasoning parameter at all, whatever the session picked, and its +request stays byte-identical to the one sent before this module existed +(pinned by the ``unruled_*`` golden payload snapshots under tests/llm/golden/). + +Choices +------- +Each chat session stores a ``ReasoningChoice`` (``Session.reasoning_effort``, +picked in the chat input), following the pi agent harness: a concrete level, +``off``, or ``provider_default`` (no reasoning parameter, so the provider's +own default applies; on a backend that always needs a level, its documented +default level is sent). A new session starts at the model's default level +(see "The default level"), which the picker marks as "Default". + +A model offers its levels, plus ``off`` when it can stop reasoning. A stored +choice the current model lacks is clamped the way pi does it: the nearest +available choice at or above the request, else the nearest below, along +``CHOICE_LADDER``. The session keeps the REQUESTED choice, so it re-applies +after a model switch. + +The default level +----------------- ``ReasoningRule.levels`` lists the levels that turn reasoning on, weakest -first. "Off" values (``none``, ``disabled``, ``minimal``) and values a -provider silently maps onto another level are left out, so the tuple holds -only levels that behave differently. The default level is: +first. Values a provider silently maps onto another level are left out, so +the tuple holds only levels that behave differently. ``minimal`` is listed +where a model accepts it (it is selectable) but is never the default. The +default level is: 1. one level below the model's strongest level (``LEVELS_BELOW_MAX``); 2. never stronger than ``LEVEL_CEILING`` (``high``); @@ -31,15 +47,15 @@ level is used instead, so a default is never lowered. Token-budget models (Claude 4.5, Gemini 2.5) have no named levels; their -rows name four budget rungs ``BUDGET_RUNGS`` so the same rule picks the +rows name four budget rungs ``BUDGET_RUNGS`` so the same rules pick the third rung ("high"). Adding a model -------------- Add its exact id (every alias and dated snapshot the provider accepts) to a -rule below, with the ordered levels from the provider's documentation and -the documentation URL in ``source``. The unit tests in -tests/llm/test_reasoning_rules.py validate every row. +rule below, with the ordered levels, default, and off behavior from the +provider's documentation and the documentation URL in ``source``. The unit +tests in tests/llm/test_reasoning_rules.py validate every row. The keys follow the ``/`` shape of the model catalog planned in docs/PROVIDER_LAYER_CATCHUP.md section 10, so the rows can move @@ -52,17 +68,48 @@ from enum import Enum from typing import Dict, Mapping, Optional, Tuple +# ─────────────────────────────── choices ────────────────────────────── + + +class ReasoningChoice(str, Enum): + """What a chat session asks for (stored as ``Session.reasoning_effort``).""" + + PROVIDER_DEFAULT = "provider_default" + OFF = "off" + MINIMAL = "minimal" + LOW = "low" + MEDIUM = "medium" + HIGH = "high" + XHIGH = "xhigh" + MAX = "max" + + +#: Clamping ladder, weakest first (pi's EXTENDED_THINKING_LEVELS). +CHOICE_LADDER: Tuple[ReasoningChoice, ...] = ( + ReasoningChoice.OFF, + ReasoningChoice.MINIMAL, + ReasoningChoice.LOW, + ReasoningChoice.MEDIUM, + ReasoningChoice.HIGH, + ReasoningChoice.XHIGH, + ReasoningChoice.MAX, +) + # ─────────────────────────────── policy ─────────────────────────────── #: How many levels below the model's strongest level the default sits. LEVELS_BELOW_MAX = 1 -#: The strongest level ever chosen as a default (unless the provider's own -#: default is already stronger; see rule 3 in the module docstring). +#: The strongest default level (unless the provider's own default is already +#: stronger; see rule 3 in the module docstring). It is also where a session +#: starts when it is created while the model in use has no rule. LEVEL_CEILING = "high" -#: Every level name any provider uses, weakest first. Rows must use these. -KNOWN_LEVELS: Tuple[str, ...] = ("minimal", "low", "medium", "high", "xhigh", "max") +#: Levels that are never the default: "minimal" barely reasons. +DEFAULT_EXCLUDED_LEVELS = frozenset({"minimal"}) + +#: Levels whose thinking needs the rule's extended output cap. +EXTENDED_LEVELS = frozenset({"xhigh", "max"}) #: Rung names for token-budget models, weakest first. BUDGET_RUNGS: Tuple[str, ...] = ("low", "medium", "high", "max") @@ -73,22 +120,25 @@ class ReasoningWire(str, Enum): - """How a reasoning default is expressed in a request.""" + """How reasoning is expressed in a request.""" - #: Top-level ``reasoning_effort: `` on a Chat Completions request. - #: The ChatGPT-subscription translator maps it to ``reasoning.effort``. + #: Top-level ``reasoning_effort: `` on a Chat Completions request + #: (off: ``"none"``). The ChatGPT-subscription translator maps it to + #: ``reasoning.effort``. EFFORT = "effort" #: Anthropic ``thinking: {"type": "adaptive"}`` plus - #: ``output_config: {"effort": }`` (Claude 4.6 and later). + #: ``output_config: {"effort": }`` (Claude 4.6 and later; off: + #: ``thinking: {"type": "disabled"}``). ANTHROPIC_ADAPTIVE = "anthropic_adaptive" #: Anthropic ``thinking: {"type": "enabled", "budget_tokens": }`` #: (Claude 4.5 generation). ANTHROPIC_BUDGET = "anthropic_budget" #: Gemini ``thinkingConfig: {"thinkingLevel": }`` (Gemini 3.x). GEMINI_LEVEL = "gemini_level" - #: Gemini ``thinkingConfig: {"thinkingBudget": }`` (Gemini 2.5). + #: Gemini ``thinkingConfig: {"thinkingBudget": }`` (Gemini 2.5; off: + #: budget 0). GEMINI_BUDGET = "gemini_budget" - #: OpenRouter ``reasoning: {"effort": }``. + #: OpenRouter ``reasoning: {"effort": }`` (off: ``"none"``). OPENROUTER_EFFORT = "openrouter_effort" #: OpenRouter ``reasoning: {"max_tokens": }``. OPENROUTER_BUDGET = "openrouter_budget" @@ -104,41 +154,92 @@ class ReasoningWire(str, Enum): ) +class ReasoningOff(str, Enum): + """How a model stops reasoning (a rule without one cannot).""" + + #: The provider's default is no reasoning: off sends nothing. + OMIT = "omit" + #: Off sends the wire's explicit disable form. + EXPLICIT = "explicit" + + +class TemperaturePolicy(str, Enum): + """When a model rejects an explicit ``temperature``.""" + + KEEP = "keep" + OMIT_WHILE_THINKING = "omit_while_thinking" + OMIT_ALWAYS = "omit_always" + + @dataclass(frozen=True) class ReasoningRule: """Reasoning capabilities of one model (or of models that share them).""" - #: How the default is expressed in the request. + #: How reasoning is expressed in the request. wire: ReasoningWire #: Levels that turn reasoning on, weakest first (``BUDGET_RUNGS`` for #: budget wires). levels: Tuple[str, ...] #: Level the provider applies when the parameter is omitted, when that - #: is one of ``levels``. None when omitting it means no reasoning, a - #: provider-chosen dynamic amount, or an "off" value. + #: is one of ``levels``. None when omitting it means no reasoning or a + #: provider-chosen (dynamic or undocumented) amount. provider_default: Optional[str] #: Documentation the row was taken from. source: str #: Budget-wire rows: token budget per rung, aligned with ``levels``. budgets: Tuple[int, ...] = () - #: Output-token cap to request while reasoning is on. Reasoning tokens - #: count against the provider's output cap, so it must leave room for - #: the answer. 0 keeps the caller's cap unchanged. + #: Output-token cap to request while reasoning. Reasoning tokens count + #: against the provider's output cap, so it must leave room for the + #: answer. 0 keeps the caller's cap unchanged. output_tokens: int = 0 - #: The model rejects ``temperature`` while reasoning is on. - omit_temperature: bool = False + #: Output-token cap at ``EXTENDED_LEVELS``; 0 means ``output_tokens``. + extended_output_tokens: int = 0 + #: How the model stops reasoning; None when it cannot. + off: Optional[ReasoningOff] = None + #: When the model rejects an explicit temperature. + temperature: TemperaturePolicy = TemperaturePolicy.KEEP + #: The backend requires a level on every request (the Codex backend), so + #: ``provider_default`` sends the documented default level explicitly. + requires_level: bool = False + + def ladder_choices(self) -> Tuple[ReasoningChoice, ...]: + """The ladder choices this model accepts, weakest first.""" + return tuple( + choice + for choice in CHOICE_LADDER + if (choice is ReasoningChoice.OFF and self.off is not None) + or choice.value in self.levels + ) + + def output_tokens_for(self, level: Optional[str]) -> int: + """Output cap while reasoning at ``level`` (None: the provider's).""" + if level in EXTENDED_LEVELS and self.extended_output_tokens: + return self.extended_output_tokens + return self.output_tokens + + def budget_for(self, level: str) -> int: + """Token budget of a budget-wire rung.""" + return self.budgets[self.levels.index(level)] @dataclass(frozen=True) class ReasoningDecision: - """The reasoning default resolved for one (provider, model) pair.""" + """What one request sends, resolved from a model's rule and a choice.""" #: ``/`` of the row that produced this decision. key: str + #: The choice the session asked for. + requested: ReasoningChoice + #: The choice in effect after clamping it to this model. + choice: ReasoningChoice wire: ReasoningWire - level: str + #: Level sent, or None when no level is sent (off, provider default). + level: Optional[str] #: Token budget for budget wires, else None. budget_tokens: Optional[int] + #: Send the wire's explicit disable form. + off: bool + #: Output cap the request needs; 0 keeps the caller's cap. output_tokens: int omit_temperature: bool @@ -151,20 +252,69 @@ def output_cap(self, base: int) -> int: def describe(self) -> str: """One-line human-readable summary for logs.""" - if self.budget_tokens is not None: - return f"{self.wire.value} budget={self.budget_tokens} ({self.level})" - return f"{self.wire.value} level={self.level}" + choice = self.requested.value + if self.choice is not self.requested: + choice = f"{choice} (clamped to {self.choice.value})" + if self.off: + sent = "disable" + elif self.level is None and self.choice is ReasoningChoice.OFF: + sent = "nothing (the model does not reason by default)" + elif self.level is None: + sent = "nothing (provider default)" + elif self.budget_tokens is not None: + sent = f"budget={self.budget_tokens} ({self.level})" + else: + sent = f"level={self.level}" + return f"choice={choice}, {self.wire.value} {sent}" + + +@dataclass(frozen=True) +class ReasoningOptions: + """What the chat-input picker offers for one model.""" + + #: Selectable choices, in display order. + choices: Tuple[ReasoningChoice, ...] + #: The model's default level (marked "Default" in the picker). + default_level: str + #: What ``provider_default`` does: "off", a level, or "dynamic". + provider_default: str + #: Effective choice for every possible stored choice. + resolution: Mapping[ReasoningChoice, ReasoningChoice] + + +def default_level(rule: ReasoningRule) -> str: + """The default level for ``rule`` (see the module docstring).""" + ladder = tuple( + level for level in rule.levels if level not in DEFAULT_EXCLUDED_LEVELS + ) + index = max(len(ladder) - 1 - LEVELS_BELOW_MAX, 0) + if LEVEL_CEILING in ladder: + index = min(index, ladder.index(LEVEL_CEILING)) + if rule.provider_default in ladder and index < ladder.index(rule.provider_default): + index = len(ladder) - 1 + return ladder[index] -def target_level(rule: ReasoningRule) -> str: - """Pick the default level for ``rule`` (see the module docstring).""" - levels = rule.levels - index = max(len(levels) - 1 - LEVELS_BELOW_MAX, 0) - if LEVEL_CEILING in levels: - index = min(index, levels.index(LEVEL_CEILING)) - if rule.provider_default in levels and index < levels.index(rule.provider_default): - index = len(levels) - 1 - return levels[index] +def clamp_choice(rule: ReasoningRule, choice: ReasoningChoice) -> ReasoningChoice: + """The choice that takes effect for ``rule`` when a session asks for ``choice``. + + Ladder choices the model lacks move to the nearest available choice at or + above the request, else the nearest below (pi's ``clampThinkingLevel``). + """ + if choice is ReasoningChoice.PROVIDER_DEFAULT: + # Omitting the parameter IS off on these models. + return ReasoningChoice.OFF if rule.off is ReasoningOff.OMIT else choice + available = rule.ladder_choices() + if choice in available: + return choice + requested = CHOICE_LADDER.index(choice) + for candidate in CHOICE_LADDER[requested:]: + if candidate in available: + return candidate + for candidate in reversed(CHOICE_LADDER[:requested]): + if candidate in available: + return candidate + raise ValueError(f"Reasoning rule has no levels: {rule!r}") def reasoning_surface(provider: str, auth_mode: str) -> str: @@ -178,29 +328,93 @@ def reasoning_surface(provider: str, auth_mode: str) -> str: return provider +def _lookup( + provider: Optional[str], model: Optional[str], auth_mode: str +) -> Tuple[Optional[str], Optional[ReasoningRule]]: + if not provider or not model: + return None, None + surface = reasoning_surface(provider, auth_mode) + return f"{surface}/{model}", REASONING_RULES.get(surface, {}).get(model) + + +def default_choice( + provider: Optional[str], model: Optional[str], auth_mode: str = "api_key" +) -> ReasoningChoice: + """The choice a new session starts at with this model in use. + + The model's default level; ``LEVEL_CEILING`` when the model has no rule + (the session then takes a level that most models offer, clamped to the + next model with a rule). + """ + _, rule = _lookup(provider, model, auth_mode) + return ReasoningChoice(default_level(rule) if rule is not None else LEVEL_CEILING) + + def resolve_reasoning( provider: Optional[str], model: Optional[str], - auth_mode: str = "api_key", + auth_mode: str, + choice: ReasoningChoice, ) -> Optional[ReasoningDecision]: - """Resolve the reasoning default for a model, or None if it has no row.""" - if not provider or not model: - return None - surface = reasoning_surface(provider, auth_mode) - rule = REASONING_RULES.get(surface, {}).get(model) + """Resolve what a request sends for ``choice``, or None if the model has + no row.""" + key, rule = _lookup(provider, model, auth_mode) if rule is None: return None - level = target_level(rule) - budget = ( - rule.budgets[rule.levels.index(level)] if rule.wire in BUDGET_WIRES else None - ) + effective = clamp_choice(rule, choice) + if effective is ReasoningChoice.PROVIDER_DEFAULT: + level = rule.provider_default if rule.requires_level else None + elif effective is ReasoningChoice.OFF: + level = None + else: + level = effective.value + + if level is not None: + thinking = True + elif effective is ReasoningChoice.PROVIDER_DEFAULT: + thinking = rule.off is not ReasoningOff.OMIT + else: + thinking = False + return ReasoningDecision( - key=f"{surface}/{model}", + key=key, + requested=choice, + choice=effective, wire=rule.wire, level=level, - budget_tokens=budget, - output_tokens=rule.output_tokens, - omit_temperature=rule.omit_temperature, + budget_tokens=( + rule.budget_for(level) + if level is not None and rule.wire in BUDGET_WIRES + else None + ), + off=effective is ReasoningChoice.OFF and rule.off is ReasoningOff.EXPLICIT, + output_tokens=rule.output_tokens_for(level) if thinking else 0, + omit_temperature=( + rule.temperature is TemperaturePolicy.OMIT_ALWAYS + or (rule.temperature is TemperaturePolicy.OMIT_WHILE_THINKING and thinking) + ), + ) + + +def reasoning_options( + provider: Optional[str], model: Optional[str], auth_mode: str +) -> Optional[ReasoningOptions]: + """What the picker offers for a model, or None if the model has no row.""" + _, rule = _lookup(provider, model, auth_mode) + if rule is None: + return None + if rule.off is ReasoningOff.OMIT: + # Omitting the parameter is off: provider default and off coincide. + meta: Tuple[ReasoningChoice, ...] = () + provider_default = ReasoningChoice.OFF.value + else: + meta = (ReasoningChoice.PROVIDER_DEFAULT,) + provider_default = rule.provider_default or "dynamic" + return ReasoningOptions( + choices=meta + rule.ladder_choices(), + default_level=default_level(rule), + provider_default=provider_default, + resolution={choice: clamp_choice(rule, choice) for choice in ReasoningChoice}, ) @@ -240,28 +454,34 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: # Output caps. Non-streaming Anthropic requests are refused by the Anthropic # SDK above 21,333 max_tokens (anthropic/_base_client.py # _calculate_nonstreaming_timeout), so Anthropic-family rows stay just below -# it. OpenAI-style effort rows use 32,000: OpenAI recommends reserving at -# least 25,000 tokens for reasoning plus output -# (https://developers.openai.com/api/docs/guides/reasoning). +# it at every level. OpenAI-style effort rows use 32,000 (OpenAI recommends +# reserving at least 25,000 tokens for reasoning plus output, +# https://developers.openai.com/api/docs/guides/reasoning) and 64,000 at +# xhigh/max, where thinking alone can pass 32,000; every such model's +# documented output limit is above that. _ANTHROPIC_OUTPUT_TOKENS = 21_000 _EFFORT_OUTPUT_TOKENS = 32_000 +_EFFORT_EXTENDED_OUTPUT_TOKENS = 64_000 # ── OpenAI public API (Chat Completions ``reasoning_effort``) ── # Values and defaults: https://developers.openai.com/api/docs/models/. -# Models before gpt-5.1 reject "none"; gpt-5.1+ reject "minimal"; xhigh -# arrived with gpt-5.2 and max with gpt-5.6. Temperature is rejected whenever -# effort is not "none" (the OpenAI profile already omits it). +# Models before gpt-5.1 reject "none" and never accept temperature; gpt-5.1+ +# reject "minimal", default to no reasoning through gpt-5.4, and accept +# temperature only at "none". xhigh arrived with gpt-5.2. "max" exists only on +# the Responses API: Chat Completions rejects it on gpt-5.6 and GPT-6 ("Supported +# values are: 'none', 'low', 'medium', 'high', and 'xhigh'", live API error, +# 2026-10-02; GPT-6 Astra and 6.1 Sol list the same without 'none'). _OPENAI_DOCS = "https://developers.openai.com/api/docs/models" _OPENAI_GPT5 = ReasoningRule( wire=ReasoningWire.EFFORT, - levels=("low", "medium", "high"), + levels=("minimal", "low", "medium", "high"), provider_default="medium", source=f"{_OPENAI_DOCS}/gpt-5", output_tokens=_EFFORT_OUTPUT_TOKENS, - omit_temperature=True, + temperature=TemperaturePolicy.OMIT_ALWAYS, ) _OPENAI_O_SERIES = ReasoningRule( wire=ReasoningWire.EFFORT, @@ -269,23 +489,26 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: provider_default="medium", source="https://developers.openai.com/api/docs/guides/reasoning", output_tokens=_EFFORT_OUTPUT_TOKENS, - omit_temperature=True, + temperature=TemperaturePolicy.OMIT_ALWAYS, ) _OPENAI_GPT51 = ReasoningRule( wire=ReasoningWire.EFFORT, levels=("low", "medium", "high"), - provider_default=None, # "none": no reasoning + provider_default=None, source=f"{_OPENAI_DOCS}/gpt-5.1", output_tokens=_EFFORT_OUTPUT_TOKENS, - omit_temperature=True, + off=ReasoningOff.OMIT, + temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) _OPENAI_GPT52_TO_54 = ReasoningRule( wire=ReasoningWire.EFFORT, levels=("low", "medium", "high", "xhigh"), - provider_default=None, # "none": no reasoning + provider_default=None, source=f"{_OPENAI_DOCS}/gpt-5.2", output_tokens=_EFFORT_OUTPUT_TOKENS, - omit_temperature=True, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, + off=ReasoningOff.OMIT, + temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) _OPENAI_GPT55 = ReasoningRule( wire=ReasoningWire.EFFORT, @@ -293,30 +516,41 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: provider_default="medium", source=f"{_OPENAI_DOCS}/gpt-5.5", output_tokens=_EFFORT_OUTPUT_TOKENS, - omit_temperature=True, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, + off=ReasoningOff.EXPLICIT, + temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) _OPENAI_GPT56_PLUS = ReasoningRule( wire=ReasoningWire.EFFORT, - levels=("low", "medium", "high", "xhigh", "max"), + levels=("low", "medium", "high", "xhigh"), provider_default="medium", source=f"{_OPENAI_DOCS}/gpt-6-sol", output_tokens=_EFFORT_OUTPUT_TOKENS, - omit_temperature=True, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, + off=ReasoningOff.EXPLICIT, + temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) -#: GPT-6 Astra rejects "none" and documents no default. -_OPENAI_GPT6_ASTRA = ReasoningRule( +#: GPT-6 Astra and GPT-6.1 Sol reject "none" and always reason. +_OPENAI_GPT6_ALWAYS_REASONING = ReasoningRule( wire=ReasoningWire.EFFORT, - levels=("low", "medium", "high", "xhigh", "max"), + levels=("low", "medium", "high", "xhigh"), provider_default=None, source=f"{_OPENAI_DOCS}/gpt-6-astra", output_tokens=_EFFORT_OUTPUT_TOKENS, - omit_temperature=True, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, + temperature=TemperaturePolicy.OMIT_ALWAYS, +) +_OPENAI_GPT61_SOL = replace( + _OPENAI_GPT6_ALWAYS_REASONING, + provider_default="medium", + source=f"{_OPENAI_DOCS}/gpt-6.1-sol", ) # ── ChatGPT subscription (Codex backend ``reasoning.effort``) ── -# The translator always sends a reasoning block and drops output caps, so -# output_tokens stays 0. Levels are the catalogue's supported_reasoning_levels -# minus the client-side "ultra" alias +# The translator always sends a reasoning block and drops output caps and +# temperature, so output_tokens stays 0. Levels are the catalogue's +# supported_reasoning_levels minus the client-side "ultra" alias; no model +# lists "none", so none can be turned off # (https://github.com/openai/codex/blob/main/codex-rs/models-manager/models.json). _CODEX_CATALOG = ( @@ -327,27 +561,27 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: levels=("low", "medium", "high", "xhigh"), provider_default="medium", source=_CODEX_CATALOG, + requires_level=True, ) _CODEX_MAX_LADDER_LOW_DEFAULT = ReasoningRule( wire=ReasoningWire.EFFORT, levels=("low", "medium", "high", "xhigh", "max"), provider_default="low", source=_CODEX_CATALOG, + requires_level=True, ) -_CODEX_MAX_LADDER_MEDIUM_DEFAULT = ReasoningRule( - wire=ReasoningWire.EFFORT, - levels=("low", "medium", "high", "xhigh", "max"), - provider_default="medium", - source=_CODEX_CATALOG, +_CODEX_MAX_LADDER_MEDIUM_DEFAULT = replace( + _CODEX_MAX_LADDER_LOW_DEFAULT, provider_default="medium" ) # ── OpenRouter (``reasoning`` object) ── # Levels: live GET https://openrouter.ai/api/v1/models ``reasoning`` -# (supported_efforts). Claude 4.5 slugs expose no effort selector, only a +# (supported_efforts; "mandatory" models cannot be turned off, the others +# take effort "none"). Claude 4.5 slugs expose no effort selector, only a # token budget, which must be >= 1024 and strictly below the request # max_tokens (https://openrouter.ai/docs/guides/best-practices/reasoning-tokens). -# Explicitly sent sampling parameters are forwarded upstream, where thinking -# rejects them, so every OpenRouter row drops temperature. +# Upstream Claude does not think unless asked. Explicitly sent sampling +# parameters are forwarded upstream, where thinking rejects them. _OPENROUTER_MODELS = "https://openrouter.ai/api/v1/models" _OPENROUTER_CLAUDE_BUDGET = ReasoningRule( @@ -357,115 +591,135 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: source="https://openrouter.ai/docs/guides/best-practices/reasoning-tokens", budgets=(4_096, 8_192, 12_288, 16_384), output_tokens=_ANTHROPIC_OUTPUT_TOKENS, - omit_temperature=True, + off=ReasoningOff.OMIT, + temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) -_OPENROUTER_CLAUDE_SONNET_46 = ReasoningRule( +_OPENROUTER_CLAUDE_46 = ReasoningRule( wire=ReasoningWire.OPENROUTER_EFFORT, levels=("low", "medium", "high", "max"), - provider_default="medium", + provider_default=None, source=_OPENROUTER_MODELS, output_tokens=_ANTHROPIC_OUTPUT_TOKENS, - omit_temperature=True, -) -_OPENROUTER_CLAUDE_OPUS_46 = replace( - _OPENROUTER_CLAUDE_SONNET_46, provider_default="high" + off=ReasoningOff.OMIT, + temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) _OPENROUTER_OPENAI_GPT5 = ReasoningRule( wire=ReasoningWire.OPENROUTER_EFFORT, - levels=("low", "medium", "high"), + levels=("minimal", "low", "medium", "high"), provider_default="medium", source=_OPENROUTER_MODELS, output_tokens=_EFFORT_OUTPUT_TOKENS, - omit_temperature=True, + temperature=TemperaturePolicy.OMIT_ALWAYS, ) -_OPENROUTER_OPENAI_GPT51 = replace(_OPENROUTER_OPENAI_GPT5, provider_default=None) -_OPENROUTER_OPENAI_XHIGH = ReasoningRule( +_OPENROUTER_OPENAI_GPT51 = ReasoningRule( wire=ReasoningWire.OPENROUTER_EFFORT, - levels=("low", "medium", "high", "xhigh"), - provider_default="medium", + levels=("low", "medium", "high"), + provider_default=None, source=_OPENROUTER_MODELS, output_tokens=_EFFORT_OUTPUT_TOKENS, - omit_temperature=True, + off=ReasoningOff.OMIT, + temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) -_OPENROUTER_OPENAI_MAX = ReasoningRule( +_OPENROUTER_OPENAI_XHIGH = ReasoningRule( wire=ReasoningWire.OPENROUTER_EFFORT, - levels=("low", "medium", "high", "xhigh", "max"), + levels=("low", "medium", "high", "xhigh"), provider_default="medium", source=_OPENROUTER_MODELS, output_tokens=_EFFORT_OUTPUT_TOKENS, - omit_temperature=True, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, + off=ReasoningOff.EXPLICIT, + temperature=TemperaturePolicy.OMIT_WHILE_THINKING, +) +_OPENROUTER_OPENAI_MAX = replace( + _OPENROUTER_OPENAI_XHIGH, levels=("low", "medium", "high", "xhigh", "max") +) +_OPENROUTER_OPENAI_MAX_MANDATORY = replace( + _OPENROUTER_OPENAI_MAX, off=None, temperature=TemperaturePolicy.OMIT_ALWAYS ) # ── Google Gemini (generateContent ``thinkingConfig``) ── # https://ai.google.dev/gemini-api/docs/generate-content/thinking. 2.5 models -# take a token budget only (thinkingLevel is a 400 on them); 3.x models take -# a lowercase thinkingLevel. Thinking tokens count toward maxOutputTokens -# (65,536 on every model below), so the cap leaves 8,192 tokens of answer on -# top of the chosen budget. "minimal" is excluded: it is near-off, and an -# error on 3.1-pro / 3.7-flash / 3.8-flash. +# take a token budget only (thinkingLevel is a 400 on them); flash and +# flash-lite can turn thinking off with budget 0, pro cannot. 3.x models take +# a lowercase thinkingLevel and cannot turn thinking off; "minimal" is an +# error on 3.1-pro, 3.7-flash, and 3.8-flash. Thinking tokens count toward +# maxOutputTokens (65,536 on every model below), so each cap leaves 8,192 +# tokens of answer on top of the largest budget it covers. _GEMINI_THINKING_DOCS = ( "https://ai.google.dev/gemini-api/docs/generate-content/thinking" ) -_GEMINI_OUTPUT_TOKENS = 32_768 +_GEMINI_ANSWER_TOKENS = 8_192 _GEMINI_25_PRO = ReasoningRule( wire=ReasoningWire.GEMINI_BUDGET, levels=BUDGET_RUNGS, provider_default=None, # dynamic thinking source=_GEMINI_THINKING_DOCS, budgets=(8_192, 16_384, 24_576, 32_768), - output_tokens=24_576 + 8_192, + output_tokens=24_576 + _GEMINI_ANSWER_TOKENS, + extended_output_tokens=32_768 + _GEMINI_ANSWER_TOKENS, ) _GEMINI_25_FLASH = ReasoningRule( wire=ReasoningWire.GEMINI_BUDGET, levels=BUDGET_RUNGS, - provider_default=None, # dynamic (flash) or no thinking (flash-lite) + provider_default=None, # dynamic thinking source=_GEMINI_THINKING_DOCS, budgets=(6_144, 12_288, 18_432, 24_576), - output_tokens=18_432 + 8_192, + output_tokens=18_432 + _GEMINI_ANSWER_TOKENS, + extended_output_tokens=24_576 + _GEMINI_ANSWER_TOKENS, + off=ReasoningOff.EXPLICIT, ) -_GEMINI_3_HIGH_DEFAULT = ReasoningRule( +#: flash-lite does not think unless given a budget. +_GEMINI_25_FLASH_LITE = replace(_GEMINI_25_FLASH, off=ReasoningOff.OMIT) +_GEMINI_3_OUTPUT_TOKENS = 32_768 +_GEMINI_3_WITH_MINIMAL = ReasoningRule( wire=ReasoningWire.GEMINI_LEVEL, - levels=("low", "medium", "high"), - provider_default="high", + levels=("minimal", "low", "medium", "high"), + provider_default="medium", source=_GEMINI_THINKING_DOCS, - output_tokens=_GEMINI_OUTPUT_TOKENS, + output_tokens=_GEMINI_3_OUTPUT_TOKENS, +) +_GEMINI_3_WITHOUT_MINIMAL = replace( + _GEMINI_3_WITH_MINIMAL, levels=("low", "medium", "high") ) -_GEMINI_3_MEDIUM_DEFAULT = replace(_GEMINI_3_HIGH_DEFAULT, provider_default="medium") -_GEMINI_3_MINIMAL_DEFAULT = replace(_GEMINI_3_HIGH_DEFAULT, provider_default=None) # ── xAI (Chat Completions ``reasoning_effort``) ── -# https://docs.x.ai/developers/models/. No Grok model accepts "max", and -# "none" is documented only for grok-4.3. grok-4.5 treats "xhigh" as "high", -# so its distinct levels stop at high. +# https://docs.x.ai/developers/models/. No Grok model accepts "max"; +# "none" (off) is documented only for grok-4.3. grok-4.5 treats "xhigh" as +# "high", so its distinct levels stop at high. _XAI_DOCS = "https://docs.x.ai/developers/models" +_XAI_REASONING_DOCS = "https://docs.x.ai/developers/model-capabilities/text/reasoning" _XAI_GROK_43 = ReasoningRule( wire=ReasoningWire.EFFORT, levels=("low", "medium", "high", "xhigh"), provider_default="low", source=f"{_XAI_DOCS}/grok-4.3", output_tokens=_EFFORT_OUTPUT_TOKENS, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, + off=ReasoningOff.EXPLICIT, ) _XAI_GROK_45 = ReasoningRule( wire=ReasoningWire.EFFORT, levels=("low", "medium", "high"), provider_default="high", - source="https://docs.x.ai/developers/model-capabilities/text/reasoning", + source=_XAI_REASONING_DOCS, output_tokens=_EFFORT_OUTPUT_TOKENS, ) _XAI_GROK_46_PLUS = ReasoningRule( wire=ReasoningWire.EFFORT, levels=("low", "medium", "high", "xhigh"), provider_default="high", - source="https://docs.x.ai/developers/model-capabilities/text/reasoning", + source=_XAI_REASONING_DOCS, output_tokens=_EFFORT_OUTPUT_TOKENS, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, ) # ── DeepSeek (``reasoning_effort``) ── # https://api-docs.deepseek.com/api/create-chat-completion: none/low/high/max -# (minimal maps to low, medium/xhigh to high), default high; max_tokens up to -# 393,216. Thinking mode ignores temperature without an error. +# (minimal maps to low, medium/xhigh to high), default high, "none" disables +# thinking; max_tokens up to 393,216. Thinking mode ignores temperature +# without an error. _DEEPSEEK_V4 = ReasoningRule( wire=ReasoningWire.EFFORT, @@ -473,12 +727,15 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: provider_default="high", source="https://api-docs.deepseek.com/api/create-chat-completion", output_tokens=_EFFORT_OUTPUT_TOKENS, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, + off=ReasoningOff.EXPLICIT, ) # ── Z.ai GLM (``reasoning_effort``, standard route) ── # https://docs.z.ai/api-reference/llm/chat-completion, default max, 128K -# output. glm-5.3 accepts only low/high/max; glm-5.2 maps low/medium to high -# and xhigh to max, so its distinct levels are high and max. +# output. glm-5.3 accepts only low/high/max and cannot stop thinking; +# glm-5.2 maps low/medium to high and xhigh to max (distinct levels: high and +# max), and "none" skips thinking. _GLM_DOCS = "https://docs.z.ai/api-reference/llm/chat-completion" _GLM_53 = ReasoningRule( @@ -487,6 +744,7 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: provider_default="max", source=_GLM_DOCS, output_tokens=_EFFORT_OUTPUT_TOKENS, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, ) _GLM_52 = ReasoningRule( wire=ReasoningWire.EFFORT, @@ -494,6 +752,8 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: provider_default="max", source=_GLM_DOCS, output_tokens=_EFFORT_OUTPUT_TOKENS, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, + off=ReasoningOff.EXPLICIT, ) # ── Moonshot Kimi (``reasoning_effort``) ── @@ -506,11 +766,13 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: provider_default="max", source="https://platform.kimi.ai/docs/guide/kimi-k3-quickstart", output_tokens=_EFFORT_OUTPUT_TOKENS, + extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, ) # ── gpt-oss on Groq / Cerebras (``reasoning_effort``) ── -# low/medium/high, default medium; out-of-set values are a 400. Output caps -# are clamped further by each provider profile's max_output_tokens. +# low/medium/high, default medium; always reasons; out-of-set values are a +# 400. Output caps are clamped further by each provider profile's +# max_output_tokens. _GPT_OSS_GROQ = ReasoningRule( wire=ReasoningWire.EFFORT, @@ -523,43 +785,45 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: _GPT_OSS_GROQ, source="https://inference-docs.cerebras.ai/capabilities/reasoning" ) +# ── Anthropic Claude (Messages API) ── +# https://platform.claude.com/docs/en/build-with-claude/thinking-troubleshooting: +# Claude 4.5/4.6/4.7/4.8 do not think unless asked; Sonnet 5 and Opus 5 think +# by default and accept {"type": "disabled"}; Sonnet 5.5, Opus 5.5, Fable, +# and Mythos always think. Temperature is incompatible with thinking on 4.5 +# and 4.6 and rejected outright from 4.7 on. + _CLAUDE_DOCS = "https://platform.claude.com/docs/en/build-with-claude/adaptive-thinking" _CLAUDE_BUDGET_DOCS = ( "https://platform.claude.com/docs/en/build-with-claude/extended-thinking" ) -#: Claude 4.6: adaptive thinking is off unless requested; effort has no -#: xhigh; sampling parameters are incompatible with thinking. +#: Claude 4.6: adaptive thinking off unless requested; effort has no xhigh. _CLAUDE_46 = ReasoningRule( wire=ReasoningWire.ANTHROPIC_ADAPTIVE, levels=("low", "medium", "high", "max"), - provider_default="high", - source=_CLAUDE_DOCS, - output_tokens=_ANTHROPIC_OUTPUT_TOKENS, - omit_temperature=True, -) - -#: Claude 4.7 and later: adaptive is the only thinking mode, the full effort -#: ladder exists, and non-default sampling parameters are a 400. -_CLAUDE_47_PLUS = ReasoningRule( - wire=ReasoningWire.ANTHROPIC_ADAPTIVE, - levels=("low", "medium", "high", "xhigh", "max"), - provider_default="high", + provider_default=None, source=_CLAUDE_DOCS, output_tokens=_ANTHROPIC_OUTPUT_TOKENS, - omit_temperature=True, + off=ReasoningOff.OMIT, + temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) - -#: Claude Opus 5.5: same surface, but its effort default is medium. -_CLAUDE_OPUS_55 = ReasoningRule( +#: Claude Opus 4.7 / 4.8: adaptive only, off unless requested. +_CLAUDE_47_48 = ReasoningRule( wire=ReasoningWire.ANTHROPIC_ADAPTIVE, levels=("low", "medium", "high", "xhigh", "max"), - provider_default="medium", + provider_default=None, source=_CLAUDE_DOCS, output_tokens=_ANTHROPIC_OUTPUT_TOKENS, - omit_temperature=True, + off=ReasoningOff.OMIT, + temperature=TemperaturePolicy.OMIT_ALWAYS, ) - +#: Claude Sonnet 5 / Opus 5: adaptive by default at effort high; disabling +#: is accepted at effort high or below (no effort is sent with it). +_CLAUDE_5 = replace(_CLAUDE_47_48, provider_default="high", off=ReasoningOff.EXPLICIT) +#: Claude Sonnet 5.5, Fable, Mythos: always think at effort high by default. +_CLAUDE_ALWAYS_THINKING = replace(_CLAUDE_47_48, provider_default="high", off=None) +#: Claude Opus 5.5: always thinks; its effort default is medium. +_CLAUDE_OPUS_55 = replace(_CLAUDE_ALWAYS_THINKING, provider_default="medium") #: Claude 4.5 generation: manual budget thinking only (adaptive is a 400, #: and effort is a 400 on Sonnet 4.5 / Haiku 4.5, so it is never sent). #: budget_tokens must be >= 1024 and < max_tokens. @@ -570,10 +834,10 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: source=_CLAUDE_BUDGET_DOCS, budgets=(4_096, 8_192, 12_288, 16_384), output_tokens=_ANTHROPIC_OUTPUT_TOKENS, - omit_temperature=True, + off=ReasoningOff.OMIT, + temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) - #: Claude on AWS Bedrock Converse takes the same thinking/effort keys through #: additionalModelRequestFields, with the same per-model rules #: (https://docs.aws.amazon.com/bedrock/latest/userguide/claude-messages-adaptive-thinking.html, @@ -584,7 +848,9 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: "claude-messages-adaptive-thinking.html" ) _BEDROCK_CLAUDE_46 = replace(_CLAUDE_46, source=_BEDROCK_DOCS) -_BEDROCK_CLAUDE_47_PLUS = replace(_CLAUDE_47_PLUS, source=_BEDROCK_DOCS) +_BEDROCK_CLAUDE_47_48 = replace(_CLAUDE_47_48, source=_BEDROCK_DOCS) +_BEDROCK_CLAUDE_5 = replace(_CLAUDE_5, source=_BEDROCK_DOCS) +_BEDROCK_CLAUDE_ALWAYS_THINKING = replace(_CLAUDE_ALWAYS_THINKING, source=_BEDROCK_DOCS) _BEDROCK_CLAUDE_OPUS_55 = replace(_CLAUDE_OPUS_55, source=_BEDROCK_DOCS) _BEDROCK_CLAUDE_45_BUDGET = replace( _CLAUDE_45_BUDGET, @@ -636,9 +902,9 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: "gpt-5.6-luna", "gpt-6-sol", "gpt-6-luna", - "gpt-6.1-sol", ), - _models(_OPENAI_GPT6_ASTRA, "gpt-6-astra"), + _models(_OPENAI_GPT6_ALWAYS_REASONING, "gpt-6-astra"), + _models(_OPENAI_GPT61_SOL, "gpt-6.1-sol"), ), "openai_subscription": _surface( _models(_CODEX_GPT55, "gpt-5.5"), @@ -660,8 +926,11 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: "anthropic/claude-haiku-4.5", "anthropic/claude-opus-4.5", ), - _models(_OPENROUTER_CLAUDE_SONNET_46, "anthropic/claude-sonnet-4.6"), - _models(_OPENROUTER_CLAUDE_OPUS_46, "anthropic/claude-opus-4.6"), + _models( + _OPENROUTER_CLAUDE_46, + "anthropic/claude-sonnet-4.6", + "anthropic/claude-opus-4.6", + ), _models(_OPENROUTER_OPENAI_GPT5, "openai/gpt-5", "openai/gpt-5-mini"), _models(_OPENROUTER_OPENAI_GPT51, "openai/gpt-5.1"), _models( @@ -677,28 +946,32 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: "openai/gpt-5.6-luna", "openai/gpt-6-sol", "openai/gpt-6-luna", + ), + _models( + _OPENROUTER_OPENAI_MAX_MANDATORY, "openai/gpt-6-astra", "openai/gpt-6.1-sol", ), ), "gemini": _surface( _models(_GEMINI_25_PRO, "gemini-2.5-pro"), - _models(_GEMINI_25_FLASH, "gemini-2.5-flash", "gemini-2.5-flash-lite"), + _models(_GEMINI_25_FLASH, "gemini-2.5-flash"), + _models(_GEMINI_25_FLASH_LITE, "gemini-2.5-flash-lite"), _models( - _GEMINI_3_HIGH_DEFAULT, + replace(_GEMINI_3_WITH_MINIMAL, provider_default="high"), "gemini-3-flash-preview", - "gemini-3.1-pro-preview", - "gemini-3.1-pro-preview-customtools", ), _models( - _GEMINI_3_MEDIUM_DEFAULT, - "gemini-3.5-flash", - "gemini-3.6-flash", - "gemini-3.7-flash", - "gemini-3.8-flash", + replace(_GEMINI_3_WITHOUT_MINIMAL, provider_default="high"), + "gemini-3.1-pro-preview", + "gemini-3.1-pro-preview-customtools", ), + _models(_GEMINI_3_WITH_MINIMAL, "gemini-3.5-flash", "gemini-3.6-flash"), + _models(_GEMINI_3_WITHOUT_MINIMAL, "gemini-3.7-flash", "gemini-3.8-flash"), _models( - _GEMINI_3_MINIMAL_DEFAULT, "gemini-3.1-flash-lite", "gemini-3.5-flash-lite" + replace(_GEMINI_3_WITH_MINIMAL, provider_default="minimal"), + "gemini-3.1-flash-lite", + "gemini-3.5-flash-lite", ), ), "grok": _surface( @@ -735,13 +1008,11 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: ), "anthropic": _surface( _models(_CLAUDE_46, "claude-sonnet-4-6", "claude-opus-4-6"), + _models(_CLAUDE_47_48, "claude-opus-4-7", "claude-opus-4-8"), + _models(_CLAUDE_5, "claude-sonnet-5", "claude-opus-5"), _models( - _CLAUDE_47_PLUS, - "claude-opus-4-7", - "claude-opus-4-8", - "claude-sonnet-5", + _CLAUDE_ALWAYS_THINKING, "claude-sonnet-5-5", - "claude-opus-5", "claude-fable-5", "claude-fable-5-1", "claude-mythos-5", @@ -790,17 +1061,23 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: *_bedrock_ids("anthropic.claude-opus-4-6-v1", "us", "eu", "au", "global"), ), _models( - _BEDROCK_CLAUDE_47_PLUS, + _BEDROCK_CLAUDE_47_48, *_bedrock_ids( "anthropic.claude-opus-4-7", "us", "eu", "jp", "au", "global" ), *_bedrock_ids( "anthropic.claude-opus-4-8", "us", "eu", "jp", "au", "global" ), + ), + _models( + _BEDROCK_CLAUDE_5, *_bedrock_ids( "anthropic.claude-sonnet-5", "us", "eu", "au", "in", "global" ), *_bedrock_ids("anthropic.claude-opus-5", "us", "eu", "au", "in", "global"), + ), + _models( + _BEDROCK_CLAUDE_ALWAYS_THINKING, *_bedrock_ids("anthropic.claude-fable-5", "us", "global"), *_bedrock_ids("anthropic.claude-fable-5-1", "us", "global"), ), diff --git a/agent_core/core/session/session.py b/agent_core/core/session/session.py index 0ef00b6f..3fde370c 100644 --- a/agent_core/core/session/session.py +++ b/agent_core/core/session/session.py @@ -16,6 +16,7 @@ from datetime import datetime from typing import List, Dict, Any, Optional +from agent_core.core.models.reasoning import ReasoningChoice from agent_core.core.session.todo import TodoItem @@ -34,6 +35,15 @@ class SessionType: MAIN_SESSION_ID = "main" +def _stored_reasoning_effort(value: Any) -> Optional[str]: + """A persisted reasoning choice, or None when absent or no longer valid. + + A None is reseeded with the model's default level by the session manager + when the session is restored. + """ + return value if value in {choice.value for choice in ReasoningChoice} else None + + @dataclass class Session: """ @@ -54,6 +64,10 @@ class Session: workspace_dir: Persistent scratch directory for this session. agent_app_project_id: Backing project id for agent_app sessions. gui_mode: Whether this session drives the GUI action space. + reasoning_effort: The session's reasoning choice (a ReasoningChoice + value, picked in the chat input). The session manager seeds it + with the model's default level when the session is created or + restored without a valid one; None only until then. action_count/token_count: Budget counters for the current run (reset when a new run starts). input_tokens/output_tokens/cache_tokens: LLM usage breakdown for @@ -79,6 +93,7 @@ class Session: workspace_dir: Optional[str] = None agent_app_project_id: Optional[str] = None gui_mode: bool = False + reasoning_effort: Optional[str] = None # Per-run budget counters action_count: int = 0 token_count: int = 0 @@ -140,6 +155,7 @@ def to_dict(self) -> Dict[str, Any]: "workspace_dir": self.workspace_dir, "agent_app_project_id": self.agent_app_project_id, "gui_mode": self.gui_mode, + "reasoning_effort": self.reasoning_effort, "action_count": self.action_count, "token_count": self.token_count, "input_tokens": self.input_tokens, @@ -168,6 +184,7 @@ def from_dict(cls, data: Dict[str, Any]) -> "Session": workspace_dir=data.get("workspace_dir"), agent_app_project_id=data.get("agent_app_project_id"), gui_mode=data.get("gui_mode", False), + reasoning_effort=_stored_reasoning_effort(data.get("reasoning_effort")), action_count=data.get("action_count", 0), token_count=data.get("token_count", 0), input_tokens=data.get("input_tokens", 0), diff --git a/agent_core/core/state/session.py b/agent_core/core/state/session.py index 3c51974a..156d19e7 100644 --- a/agent_core/core/state/session.py +++ b/agent_core/core/state/session.py @@ -20,12 +20,20 @@ # At session deletion: StateSession.end(session_id) + + # Around work done for a session (its serial loop, or a call made for it + # outside the loop), so context-scoped consumers can find it: + with StateSession.bind(session_id): + ... + state = StateSession.bound() # None outside any bound session """ from __future__ import annotations +from contextlib import contextmanager +from contextvars import ContextVar from dataclasses import dataclass, field -from typing import ClassVar, Optional, Dict, Any, TYPE_CHECKING +from typing import ClassVar, Iterator, Optional, Dict, Any, TYPE_CHECKING from agent_core.core.state.types import AgentProperties @@ -33,6 +41,14 @@ from agent_core.core.session.session import Session +#: Id of the session whose work runs in the current context. asyncio tasks +#: and asyncio.to_thread inherit it, so everything a session's loop starts +#: (actions, sub-agents, LLM calls) sees the session that started it. +_bound_session_id: ContextVar[Optional[str]] = ContextVar( + "bound_session_id", default=None +) + + @dataclass class StateSession: """Per-session runtime state isolated from other concurrent sessions. @@ -128,6 +144,21 @@ def end(cls, session_id: str) -> None: """Remove a session's state (session deletion).""" cls._instances.pop(session_id, None) + @classmethod + @contextmanager + def bind(cls, session_id: str) -> Iterator[None]: + """Mark the current context as doing work for ``session_id``.""" + token = _bound_session_id.set(session_id) + try: + yield + finally: + _bound_session_id.reset(token) + + @classmethod + def bound(cls) -> Optional["StateSession"]: + """State of the session the current context works for, if any.""" + return cls.get_or_none(_bound_session_id.get()) + @classmethod def get_all_session_ids(cls) -> list[str]: """Get all active session IDs.""" diff --git a/app/agent_base.py b/app/agent_base.py index e6157d20..9b2cc7f2 100644 --- a/app/agent_base.py +++ b/app/agent_base.py @@ -28,6 +28,7 @@ import traceback import time import json +from contextlib import nullcontext from dataclasses import dataclass from typing import Awaitable, Callable, Dict, Iterable, Optional @@ -86,6 +87,7 @@ MemoryFileWatcher, LLMCallType, ) +from agent_core.core.models.reasoning import ReasoningChoice from agent_core.core.session import Session, SessionType, MAIN_SESSION_ID from agent_core.core.state.session import StateSession from app.context_engine import ContextEngine @@ -502,10 +504,21 @@ def get_commands(self) -> Dict[str, AgentCommand]: # Session API (sidebar surface) # ===================================== - def create_chat_session(self, title: str = "New chat") -> Session: - """Create a fresh chat session (the "+ New Chat" button).""" + def create_chat_session( + self, + title: str = "New chat", + reasoning_effort: Optional[ReasoningChoice] = None, + ) -> Session: + """Create a fresh chat session (the "+ New Chat" button). + + ``reasoning_effort`` carries the draft chat's picker value into the + session the draft becomes; None (picker untouched) starts it at the + default level of the model in use. + """ return self.session_manager.create_session( - session_type=SessionType.CHAT, title=title + session_type=SessionType.CHAT, + title=title, + reasoning_effort=reasoning_effort, ) async def delete_session(self, session_id: str) -> bool: @@ -528,6 +541,12 @@ def rename_session(self, session_id: str, title: str) -> bool: """Rename a session's sidebar title.""" return self.session_manager.rename_session(session_id, title) + def set_session_reasoning_effort( + self, session_id: str, choice: ReasoningChoice + ) -> bool: + """Set a session's reasoning choice (the chat input's picker).""" + return self.session_manager.set_reasoning_effort(session_id, choice) + # ===================================== # Main Agent Cycle # ===================================== @@ -1625,15 +1644,18 @@ async def _auto_title_session( title = "" try: - response = await self.llm.generate_response_async( - system_prompt=( - "Generate a concise 2-5 word title for a conversation " - "that starts with the user request below. Reply with a " - 'JSON object: {"title": ""}. Same language ' - "as the request, no punctuation at the end." - ), - user_prompt=basis[:2000], - ) + # Titling is work for this session: it reasons as the session's + # picker says, also when started outside the session's loop. + with StateSession.bind(session_id): + response = await self.llm.generate_response_async( + system_prompt=( + "Generate a concise 2-5 word title for a conversation " + "that starts with the user request below. Reply with a " + 'JSON object: {"title": ""}. Same language ' + "as the request, no punctuation at the end." + ), + user_prompt=basis[:2000], + ) title = self._parse_session_title(response) except Exception as e: logger.debug(f"[SESSION] Auto-title LLM call failed for {session_id}: {e}") @@ -2597,15 +2619,31 @@ async def _handle_external_event(self, payload: Dict) -> None: except Exception as e: logger.error(f"Error handling external event: {e}", exc_info=True) - async def _handle_prompt_enhance(self, user_message: str) -> str: + async def _handle_prompt_enhance( + self, + user_message: str, + session_id: Optional[str], + reasoning_choice: Optional[ReasoningChoice], + ) -> str: + """Rewrite a prompt typed in the chat input for clarity. + + The call reasons like the chat the prompt is typed in: the session + ``session_id``, or, for a draft chat that has no session yet + (``session_id`` None), its picker value ``reasoning_choice``. + """ try: from agent_core.core.prompts.reasoning import ( PROMPT_ENHANCE_REASONING_PROMPT, ) - response = await self.llm.generate_response_async( - system_prompt=PROMPT_ENHANCE_REASONING_PROMPT, user_prompt=user_message - ) + with ( + StateSession.bind(session_id) if session_id is not None else nullcontext() + ): + response = await self.llm.generate_response_async( + system_prompt=PROMPT_ENHANCE_REASONING_PROMPT, + user_prompt=user_message, + reasoning_choice=reasoning_choice, + ) result = json.loads(response) return result.get("enhanced_prompt", "") except Exception as e: diff --git a/app/triggers/runtime.py b/app/triggers/runtime.py index 48e44478..68e8febb 100644 --- a/app/triggers/runtime.py +++ b/app/triggers/runtime.py @@ -26,6 +26,7 @@ QueueClosed, ) from agent_core.core.session import MAIN_SESSION_ID +from agent_core.core.state.session import StateSession from app.triggers.sources import TriggerSource if TYPE_CHECKING: @@ -409,7 +410,10 @@ async def _consume(self, session_id: str, queue: SessionTriggerQueue) -> None: ) logger.info(f"[SessionRuntime] Loop started for session {session_id}") - with session_ctx: + # Bind the session for everything its turns run (actions, sub-agents, + # LLM calls): the LLM layer reads the session's reasoning choice + # through this binding. + with session_ctx, StateSession.bind(session_id): while self._running: try: trig = await queue.get() diff --git a/app/ui_layer/adapters/browser_adapter.py b/app/ui_layer/adapters/browser_adapter.py index 2a171d2a..ba33a1ce 100644 --- a/app/ui_layer/adapters/browser_adapter.py +++ b/app/ui_layer/adapters/browser_adapter.py @@ -24,6 +24,7 @@ SCHEDULE_HOUR_DEFAULT, SCHEDULE_MINUTE_DEFAULT, ) +from agent_core.core.models.reasoning import ReasoningChoice from agent_core.utils.logger import logger from app.config import AGENT_WORKSPACE_ROOT, APP_DATA_PATH from app.i18n import tui @@ -1005,11 +1006,20 @@ async def submit_message( client_id=client_id, ) - async def _handle_enhance_prompt(self, content: str, ws) -> None: - """Enhance a user's prompt using the LLM for clarity and precision.""" + async def _handle_enhance_prompt(self, data: Dict[str, Any], ws) -> None: + """Enhance a user's prompt using the LLM for clarity and precision. + + The call reasons like the chat the prompt is typed in: its session, + or a draft chat's picker value. + """ + session_id = data.get("sessionId") or "main" try: enhanced: str = await self._controller.handle_prompt_enhance( - user_message=content + user_message=data["content"], + session_id=None if session_id == "new" else session_id, + reasoning_choice=( + self._draft_reasoning(data) if session_id == "new" else None + ), ) await self._send_to( ws, {"type": "prompt_enhanced", "content": enhanced.strip()} @@ -1451,7 +1461,9 @@ async def _handle_ws_message(self, data: Dict[str, Any], ws=None) -> None: # session_created is broadcast (with the sender's clientId) before # the message so the draft view can navigate to the real session. if session_id == "new": - session = self._controller.agent.create_chat_session() + session = self._controller.agent.create_chat_session( + reasoning_effort=self._draft_reasoning(data) + ) session_id = session.id await self._broadcast( { @@ -1509,7 +1521,9 @@ async def _handle_ws_message(self, data: Dict[str, Any], ws=None) -> None: name = command.strip().split()[0].lower() if command.strip() else "" cmd = self._controller.command_registry.get(name) if name else None if cmd is not None and cmd.requires_session: - session = self._controller.agent.create_chat_session() + session = self._controller.agent.create_chat_session( + reasoning_effort=self._draft_reasoning(data) + ) session_id = session.id await self._broadcast( { @@ -1530,7 +1544,7 @@ async def _handle_ws_message(self, data: Dict[str, Any], ws=None) -> None: elif msg_type == "enhance_prompt": content = data.get("content", "") if content and ws: - await self._handle_enhance_prompt(content, ws) + await self._handle_enhance_prompt(data, ws) elif msg_type == "chat_history": session_id = data.get("sessionId") or "main" @@ -1551,6 +1565,12 @@ async def _handle_ws_message(self, data: Dict[str, Any], ws=None) -> None: elif msg_type == "session_list": await self._handle_session_list(ws) + elif msg_type == "session_reasoning_set": + await self._handle_session_reasoning_set(data) + + elif msg_type == "reasoning_options_get": + await self._handle_reasoning_options_get(ws) + # File operations elif msg_type == "file_list": directory = data.get("directory", "") @@ -4270,8 +4290,16 @@ def _session_info(session) -> Dict[str, Any]: "createdAt": session.created_at, "lastActiveAt": session.last_active_at, "agentAppProjectId": session.agent_app_project_id, + "reasoningEffort": session.reasoning_effort, } + @staticmethod + def _draft_reasoning(data: Dict[str, Any]) -> Optional[ReasoningChoice]: + """A draft chat's picker value from a message; None when untouched + (the model's default level then applies).""" + value = data.get("reasoningEffort") + return ReasoningChoice(value) if value is not None else None + async def _handle_session_delete(self, data: Dict[str, Any]) -> None: """Delete a session and its chat history. The main session is permanent.""" from agent_core.core.session import MAIN_SESSION_ID @@ -4310,6 +4338,49 @@ async def _handle_session_rename(self, data: Dict[str, Any]) -> None: except Exception as e: logger.error(f"[SESSION] Rename failed for {session_id}: {e}") + async def _handle_session_reasoning_set(self, data: Dict[str, Any]) -> None: + """Set a session's reasoning choice (the chat input's picker).""" + session_id = (data.get("sessionId") or "").strip() + if not session_id: + return + try: + choice = ReasoningChoice(data.get("reasoningEffort")) + except ValueError: + logger.warning( + f"[SESSION] Unknown reasoning choice {data.get('reasoningEffort')!r} " + f"for {session_id}" + ) + return + if self._controller.agent.set_session_reasoning_effort(session_id, choice): + await self.broadcast_session_updated(session_id) + + async def _handle_reasoning_options_get(self, ws) -> None: + """Send what the reasoning picker offers for the model in use. + + Read from the live LLM interface, so it reflects the model actually + called (after any subscription substitution), not only the settings. + """ + llm = self._controller.agent.llm + options = llm.reasoning_options() + data: Dict[str, Any] = { + "success": True, + "model": llm.model, + "configurable": options is not None, + } + if options is not None: + data.update( + { + "choices": [choice.value for choice in options.choices], + "defaultLevel": options.default_level, + "providerDefault": options.provider_default, + "resolution": { + requested.value: effective.value + for requested, effective in options.resolution.items() + }, + } + ) + await self._send_to(ws, {"type": "reasoning_options_get", "data": data}) + async def _handle_session_clear(self, data: Dict[str, Any]) -> None: """Clear a session's conversation (chat + activity rows and agent-side state).""" diff --git a/app/ui_layer/browser/frontend/src/components/Chat/Chat.module.css b/app/ui_layer/browser/frontend/src/components/Chat/Chat.module.css index 30b2c6ad..8bd917dc 100644 --- a/app/ui_layer/browser/frontend/src/components/Chat/Chat.module.css +++ b/app/ui_layer/browser/frontend/src/components/Chat/Chat.module.css @@ -206,6 +206,12 @@ gap: var(--space-2); } +.controlsLeft { + display: flex; + align-items: center; + gap: var(--space-2); +} + .controlsRight { display: flex; align-items: center; diff --git a/app/ui_layer/browser/frontend/src/components/Chat/Chat.tsx b/app/ui_layer/browser/frontend/src/components/Chat/Chat.tsx index f56cf906..17da1cb6 100644 --- a/app/ui_layer/browser/frontend/src/components/Chat/Chat.tsx +++ b/app/ui_layer/browser/frontend/src/components/Chat/Chat.tsx @@ -41,6 +41,7 @@ import { selectPendingQuestions, } from '../../store/selectors/messages' import { QuestionBox } from './QuestionBox' +import { ReasoningPicker } from './ReasoningPicker' import { mergeTimeline, type TimelineEntry } from './timeline' import { selectSessionActivity } from '../../store/selectors/activity' import { selectSessionBusy, selectSessionRunState } from '../../store/selectors/agent' @@ -881,13 +882,13 @@ export function Chat({ sessionId, placeholder }: ChatProps) { const handleEnhancePrompt = useCallback(() => { if (!input.trim() || enhancing) return setEnhancing(true) - enhancePrompt(input.trim()) + enhancePrompt(input.trim(), sessionId) enhanceTimeoutRef.current = setTimeout(() => { enhanceTimeoutRef.current = null setEnhancing(false) showToast('error', t('chat:toast.enhanceTimedOut')) }, ENHANCE_TIMEOUT_MS) - }, [input, enhancing, enhancePrompt, showToast, t]) + }, [input, enhancing, enhancePrompt, sessionId, showToast, t]) useEffect(() => { return () => { @@ -1519,41 +1520,44 @@ export function Chat({ sessionId, placeholder }: ChatProps) { />
-
- - {plusOpen && ( -
- - -
- )} +
+
+ + {plusOpen && ( +
+ + +
+ )} +
+
diff --git a/app/ui_layer/browser/frontend/src/components/Chat/ReasoningPicker.module.css b/app/ui_layer/browser/frontend/src/components/Chat/ReasoningPicker.module.css new file mode 100644 index 00000000..e0fcd880 --- /dev/null +++ b/app/ui_layer/browser/frontend/src/components/Chat/ReasoningPicker.module.css @@ -0,0 +1,102 @@ +/* Reasoning picker beside the "+" button; mirrors the "+" menu styling. */ +.wrap { + position: relative; +} + +.button { + display: flex; + align-items: center; + gap: var(--space-1); + height: 30px; + padding: 0 10px; + border-radius: 15px; + background: transparent; + border: 1px solid var(--border-primary); + color: var(--text-secondary); + font-family: inherit; + font-size: var(--text-xs); + white-space: nowrap; + cursor: pointer; + transition: background var(--transition-fast), color var(--transition-fast); +} + +.button:hover:not(:disabled) { + background: var(--bg-tertiary); + color: var(--text-primary); +} + +.button:disabled { + opacity: 0.45; + cursor: not-allowed; +} + +.menu { + position: absolute; + bottom: calc(100% + 8px); + left: 0; + min-width: 220px; + padding: 4px; + background: var(--bg-secondary); + border: 1px solid var(--border-primary); + border-radius: var(--radius-md); + box-shadow: 0 4px 16px rgba(0, 0, 0, 0.5); + z-index: 999; +} + +.header { + padding: 6px 10px 4px; + color: var(--text-tertiary); + font-size: 11px; + white-space: nowrap; + overflow: hidden; + text-overflow: ellipsis; +} + +.item { + display: flex; + align-items: center; + gap: var(--space-2); + width: 100%; + padding: 7px 10px; + background: transparent; + border: none; + border-radius: var(--radius-sm); + color: var(--text-secondary); + font-family: inherit; + font-size: var(--text-sm); + cursor: pointer; + text-align: left; +} + +.item:hover { + background: var(--bg-tertiary); + color: var(--text-primary); +} + +.itemActive { + color: var(--text-primary); +} + +.check { + display: flex; + width: 14px; + flex-shrink: 0; +} + +.itemLabel { + flex: 1; +} + +.itemDetail { + color: var(--text-tertiary); + font-size: 11px; +} + +.note { + margin-top: 4px; + padding: 6px 10px 4px; + border-top: 1px solid var(--border-primary); + color: var(--text-tertiary); + font-size: 11px; + line-height: 1.4; +} diff --git a/app/ui_layer/browser/frontend/src/components/Chat/ReasoningPicker.tsx b/app/ui_layer/browser/frontend/src/components/Chat/ReasoningPicker.tsx new file mode 100644 index 00000000..3e50a23a --- /dev/null +++ b/app/ui_layer/browser/frontend/src/components/Chat/ReasoningPicker.tsx @@ -0,0 +1,159 @@ +import { useEffect, useRef, useState } from 'react' +import { useTranslation } from 'react-i18next' +import { Brain, Check, ChevronDown } from 'lucide-react' +import type { ReasoningChoice } from '../../types' +import { useWebSocket } from '../../contexts/WebSocketContext' +import { usePersistedState } from '../../hooks' +import { useAppSelector } from '../../store/hooks' +import { RESOURCES, useResource } from '../../store/resources' +import { selectReasoningOptions } from '../../store/selectors/reasoning' +import { selectSessionById } from '../../store/selectors/sessions' +import { UI_STATE } from '../../store/uiState' +import styles from './ReasoningPicker.module.css' + +const CHOICE_LABEL_KEYS = { + provider_default: 'chat:reasoning.choice.provider_default', + off: 'chat:reasoning.choice.off', + minimal: 'chat:reasoning.choice.minimal', + low: 'chat:reasoning.choice.low', + medium: 'chat:reasoning.choice.medium', + high: 'chat:reasoning.choice.high', + xhigh: 'chat:reasoning.choice.xhigh', + max: 'chat:reasoning.choice.max', +} as const satisfies Record + +const isReasoningChoice = (value: string): value is ReasoningChoice => + Object.prototype.hasOwnProperty.call(CHOICE_LABEL_KEYS, value) + +interface ReasoningPickerProps { + /** The chat this composer belongs to ('new' for the draft view). */ + sessionId: string +} + +/** + * How hard the model reasons for this chat (chat input, beside "+"). + * + * Follows the pi agent harness: each chat holds a concrete choice, and the + * menu lists only what the model in use accepts, with the model's default + * level marked "Default" (where every new chat starts). A real session's + * choice is saved server-side; the draft view keeps it in UI state until its + * first message creates the session (untouched: the default level). A stored + * choice the model lacks shows the level actually used instead (the backend + * clamps it the way pi does and keeps the stored choice for other models). + */ +export function ReasoningPicker({ sessionId }: ReasoningPickerProps) { + const { t } = useTranslation(['chat', 'common']) + const { setSessionReasoning } = useWebSocket() + useResource(RESOURCES.reasoningOptions) + const options = useAppSelector(selectReasoningOptions) + const session = useAppSelector(state => selectSessionById(state, sessionId)) + const [draftChoice] = usePersistedState(UI_STATE.chat.draftReasoningEffort) + const [open, setOpen] = useState(false) + const wrapRef = useRef(null) + + // The chat's own choice; null while it has none yet (an untouched draft, + // or the main session before the backend has sent it). + const stored: ReasoningChoice | null = + sessionId === 'new' ? draftChoice : (session?.reasoningEffort ?? null) + + useEffect(() => { + if (!open) return + const handler = (e: MouseEvent) => { + if (wrapRef.current && !wrapRef.current.contains(e.target as Node)) setOpen(false) + } + document.addEventListener('mousedown', handler) + return () => document.removeEventListener('mousedown', handler) + }, [open]) + + const choiceLabel = (value: ReasoningChoice): string => t(CHOICE_LABEL_KEYS[value]) + + // What a level-like value from the backend reads as: a choice, or the + // provider deciding the amount itself. + const levelLabel = (value: string): string => + isReasoningChoice(value) ? choiceLabel(value) : t('chat:reasoning.dynamic') + + if (!options || !options.configurable) { + return ( +
+ +
+ ) + } + + const { defaultLevel, providerDefault } = options + const requested: ReasoningChoice = stored ?? defaultLevel + const effective = options.resolution[requested] + // A choice as it takes effect on this model. + const summary = (value: ReasoningChoice): string => + value === 'provider_default' + ? t('chat:reasoning.withLevel', { + choice: choiceLabel('provider_default'), + level: levelLabel(providerDefault), + }) + : choiceLabel(value) + // The row marked as current: the requested choice, or what it resolves to + // when the model does not offer it. + const marked = options.choices.includes(requested) ? requested : effective + // Only a level the model lacks is clamped; provider_default on a model + // whose default is no reasoning simply IS off. + const clamped = !options.choices.includes(requested) && requested !== 'provider_default' + + return ( +
+ + {open && ( +
+
{t('chat:reasoning.header', { model: options.model })}
+ {options.choices.map(value => ( + + ))} + {clamped && ( +
+ {t('chat:reasoning.clamped', { requested: choiceLabel(requested), effective: summary(effective) })} +
+ )} +
+ )} +
+ ) +} diff --git a/app/ui_layer/browser/frontend/src/components/layout/NavBar.tsx b/app/ui_layer/browser/frontend/src/components/layout/NavBar.tsx index 773d09f7..ba88eabd 100644 --- a/app/ui_layer/browser/frontend/src/components/layout/NavBar.tsx +++ b/app/ui_layer/browser/frontend/src/components/layout/NavBar.tsx @@ -680,6 +680,7 @@ export function NavBar({ collapsed = false, onToggleCollapsed }: NavBarProps) { title: 'Main', createdAt: '', lastActiveAt: '', + reasoningEffort: null, } return ( diff --git a/app/ui_layer/browser/frontend/src/contexts/WebSocketContext.tsx b/app/ui_layer/browser/frontend/src/contexts/WebSocketContext.tsx index 6248a6cc..ffaba7bb 100644 --- a/app/ui_layer/browser/frontend/src/contexts/WebSocketContext.tsx +++ b/app/ui_layer/browser/frontend/src/contexts/WebSocketContext.tsx @@ -3,7 +3,7 @@ import { useNavigate } from 'react-router-dom' import { useStore } from 'react-redux' import type { ChatMessage, SessionInfo, WSMessage, MetricsTimePeriod, - AgentAppCreateRequest, + AgentAppCreateRequest, ReasoningChoice, } from '../types' import { QUESTION_DISMISSED } from '../types' import i18n from '../i18n/config' @@ -36,6 +36,8 @@ import { markStopping as agentAppMarkStopping, } from '../store/slices/agentAppSlice' import { setStatus, setSessionRunState } from '../store/slices/agentSlice' +import { upsertSession } from '../store/slices/sessionsSlice' +import { selectSessionById } from '../store/selectors/sessions' import { setUiState } from '../store/slices/uiSlice' import { selectUiState } from '../store/selectors/ui' import { UI_STATE } from '../store/uiState' @@ -114,6 +116,9 @@ interface WebSocketContextType extends WebSocketState { deleteSession: (sessionId: string) => void renameSession: (sessionId: string, title: string) => void clearSession: (sessionId: string) => void + // Reasoning picker: a session's choice (the draft view keeps it locally + // until its first message creates the session) + setSessionReasoning: (sessionId: string, choice: ReasoningChoice) => void requestChatHistory: (sessionId: string, beforeTimestamp?: number, limit?: number) => void // Per-session unread tracking (read with UI_STATE.chat.lastSeenMessageIds) markSessionSeen: (sessionId: string) => void @@ -127,8 +132,8 @@ interface WebSocketContextType extends WebSocketState { submitOnboardingStep: (value: string | string[] | Record) => void skipOnboardingStep: () => void goBackOnboardingStep: () => void - // Enhance prompt - enhancePrompt: (content: string) => void + // Enhance prompt (reasons like the chat it is typed in) + enhancePrompt: (content: string, sessionId: string) => void clearEnhancedPrompt: () => void // Local LLM (Ollama) methods checkLocalLLM: () => void @@ -240,6 +245,9 @@ export function WebSocketProvider({ children }: { children: ReactNode }) { sessionId: session.id, state: startsRun === false ? 'idle' : 'running', })) + // The session now holds the draft's reasoning choice; the next + // draft starts untouched (the model's default level) again. + dispatch(setUiState(UI_STATE.chat.draftReasoningEffort, null)) navigateRef.current(`/session/${session.id}`, { replace: true }) } break @@ -281,6 +289,19 @@ export function WebSocketProvider({ children }: { children: ReactNode }) { for (const envelope of expired) rollbackExpiredSend(envelope, dispatch) }), [dispatch, showToast]) + // A draft chat's reasoning choice rides along with whatever turns the + // draft into a session (first message, session command) or calls the LLM + // for it (prompt enhance); a real session's choice is read server-side. + // An untouched draft sends none: the model's default level applies. + const draftReasoning = useCallback( + (sessionId: string): { reasoningEffort?: ReasoningChoice } => { + if (sessionId !== 'new') return {} + const choice = selectUiState(store.getState(), UI_STATE.chat.draftReasoningEffort) + return choice === null ? {} : { reasoningEffort: choice } + }, + [store], + ) + const sendMessage = useCallback(( content: string, attachments: PendingAttachment[] | undefined, @@ -332,8 +353,9 @@ export function WebSocketProvider({ children }: { children: ReactNode }) { ), replyContext: replyContext || null, clientId, + ...draftReasoning(sessionId), })) - }, [sendOrQueue, dispatch]) + }, [sendOrQueue, dispatch, draftReasoning]) const sendCommand = useCallback((command: string, sessionId: string) => { const clientId = newClientId() @@ -348,8 +370,10 @@ export function WebSocketProvider({ children }: { children: ReactNode }) { pendingDraftClientIdsRef.current.add(clientId) } - sendOrQueue(JSON.stringify({ type: 'command', command, sessionId, clientId })) - }, [sendOrQueue]) + sendOrQueue(JSON.stringify({ + type: 'command', command, sessionId, clientId, ...draftReasoning(sessionId), + })) + }, [sendOrQueue, draftReasoning]) // Force-stop a session's in-flight run (chat input's stop button). // Optimistically enters 'stopping' so the button spins instantly; the @@ -374,6 +398,18 @@ export function WebSocketProvider({ children }: { children: ReactNode }) { sendOrQueue(JSON.stringify({ type: 'session_clear', sessionId })) }, [sendOrQueue]) + const setSessionReasoning = useCallback((sessionId: string, choice: ReasoningChoice) => { + if (sessionId === 'new') { + dispatch(setUiState(UI_STATE.chat.draftReasoningEffort, choice)) + return + } + // Optimistic: the picker reflects the choice at once; the server's + // session_updated broadcast is authoritative. + const session = selectSessionById(store.getState(), sessionId) + if (session) dispatch(upsertSession({ ...session, reasoningEffort: choice })) + sendOrQueue(JSON.stringify({ type: 'session_reasoning_set', sessionId, reasoningEffort: choice })) + }, [sendOrQueue, dispatch, store]) + const requestChatHistory = useCallback(( sessionId: string, beforeTimestamp?: number, @@ -407,9 +443,11 @@ export function WebSocketProvider({ children }: { children: ReactNode }) { dispatch(setUiState(UI_STATE.chat.lastSeenMessageIds, { ...seen, [sessionId]: lastId })) }, [dispatch, store]) - const enhancePrompt = useCallback((content: string) => { - sendOrQueue(JSON.stringify({ type: 'enhance_prompt', content })) - }, [sendOrQueue]) + const enhancePrompt = useCallback((content: string, sessionId: string) => { + sendOrQueue(JSON.stringify({ + type: 'enhance_prompt', content, sessionId, ...draftReasoning(sessionId), + })) + }, [sendOrQueue, draftReasoning]) const clearEnhancedPrompt = useCallback(() => { setState(prev => ({ ...prev, enhancedPrompt: null })) @@ -616,6 +654,7 @@ export function WebSocketProvider({ children }: { children: ReactNode }) { deleteSession, renameSession, clearSession, + setSessionReasoning, requestChatHistory, markSessionSeen, openFile, @@ -648,7 +687,7 @@ export function WebSocketProvider({ children }: { children: ReactNode }) { updateAgentAppTheme, }), [ state, sendMessage, sendCommand, stopSession, deleteSession, renameSession, clearSession, - requestChatHistory, markSessionSeen, openFile, openFolder, requestFilteredMetrics, + setSessionReasoning, requestChatHistory, markSessionSeen, openFile, openFolder, requestFilteredMetrics, subscribeDashboardMetrics, unsubscribeDashboardMetrics, requestOnboardingStep, submitOnboardingStep, skipOnboardingStep, goBackOnboardingStep, checkLocalLLM, testLocalLLMConnection, installLocalLLM, startLocalLLM, requestSuggestedModels, pullOllamaModel, diff --git a/app/ui_layer/browser/frontend/src/locales/en/chat.json b/app/ui_layer/browser/frontend/src/locales/en/chat.json index 58d1b790..e6b429ea 100644 --- a/app/ui_layer/browser/frontend/src/locales/en/chat.json +++ b/app/ui_layer/browser/frontend/src/locales/en/chat.json @@ -31,6 +31,26 @@ "stopRun": "Stop run", "stoppingRun": "Stopping run" }, + "reasoning": { + "title": "Reasoning", + "picker": "Reasoning effort", + "header": "Reasoning · {{model}}", + "withLevel": "{{choice}} · {{level}}", + "dynamic": "Model decides", + "defaultTag": "Default", + "notConfigurable": "Reasoning isn't adjustable for {{model}}", + "clamped": "{{requested}} isn't available on this model, so {{effective}} is used.", + "choice": { + "provider_default": "Provider default", + "off": "Off", + "minimal": "Minimal", + "low": "Low", + "medium": "Medium", + "high": "High", + "xhigh": "Extra high", + "max": "Max" + } + }, "reply": { "replyingTo": "Replying to: <1>{{name}}", "cancel": "Cancel reply" diff --git a/app/ui_layer/browser/frontend/src/locales/es/chat.json b/app/ui_layer/browser/frontend/src/locales/es/chat.json index 2236e221..9713e856 100644 --- a/app/ui_layer/browser/frontend/src/locales/es/chat.json +++ b/app/ui_layer/browser/frontend/src/locales/es/chat.json @@ -31,6 +31,26 @@ "stopRun": "Detener ejecución", "stoppingRun": "Deteniendo la ejecución" }, + "reasoning": { + "title": "Razonamiento", + "picker": "Nivel de razonamiento", + "header": "Razonamiento · {{model}}", + "withLevel": "{{choice}} · {{level}}", + "dynamic": "Lo decide el modelo", + "defaultTag": "Predeterminado", + "notConfigurable": "El razonamiento no es ajustable en {{model}}", + "clamped": "{{requested}} no está disponible en este modelo, así que se usa {{effective}}.", + "choice": { + "provider_default": "Predeterminado del proveedor", + "off": "Desactivado", + "minimal": "Mínimo", + "low": "Bajo", + "medium": "Medio", + "high": "Alto", + "xhigh": "Muy alto", + "max": "Máximo" + } + }, "reply": { "replyingTo": "Respondiendo a: <1>{{name}}", "cancel": "Cancelar respuesta" diff --git a/app/ui_layer/browser/frontend/src/locales/id/chat.json b/app/ui_layer/browser/frontend/src/locales/id/chat.json index 2c2ec0c2..dfe7e89e 100644 --- a/app/ui_layer/browser/frontend/src/locales/id/chat.json +++ b/app/ui_layer/browser/frontend/src/locales/id/chat.json @@ -31,6 +31,26 @@ "stopRun": "Hentikan proses", "stoppingRun": "Menghentikan proses" }, + "reasoning": { + "title": "Penalaran", + "picker": "Tingkat penalaran", + "header": "Penalaran · {{model}}", + "withLevel": "{{choice}} · {{level}}", + "dynamic": "Ditentukan model", + "defaultTag": "Bawaan", + "notConfigurable": "Penalaran tidak dapat diatur untuk {{model}}", + "clamped": "{{requested}} tidak tersedia pada model ini, jadi {{effective}} yang digunakan.", + "choice": { + "provider_default": "Bawaan penyedia", + "off": "Mati", + "minimal": "Minimal", + "low": "Rendah", + "medium": "Sedang", + "high": "Tinggi", + "xhigh": "Sangat tinggi", + "max": "Maksimum" + } + }, "reply": { "replyingTo": "Membalas: <1>{{name}}", "cancel": "Batalkan balasan" diff --git a/app/ui_layer/browser/frontend/src/locales/ja/chat.json b/app/ui_layer/browser/frontend/src/locales/ja/chat.json index b6125efe..743c3719 100644 --- a/app/ui_layer/browser/frontend/src/locales/ja/chat.json +++ b/app/ui_layer/browser/frontend/src/locales/ja/chat.json @@ -31,6 +31,26 @@ "stopRun": "実行を停止", "stoppingRun": "実行を停止中" }, + "reasoning": { + "title": "推論", + "picker": "推論の強度", + "header": "推論 · {{model}}", + "withLevel": "{{choice}} · {{level}}", + "dynamic": "モデルが決定", + "defaultTag": "既定", + "notConfigurable": "{{model}} では推論を調整できません", + "clamped": "{{requested}} はこのモデルでは使えないため、{{effective}} を使用します。", + "choice": { + "provider_default": "プロバイダーの既定", + "off": "オフ", + "minimal": "最小", + "low": "低", + "medium": "中", + "high": "高", + "xhigh": "超高", + "max": "最大" + } + }, "reply": { "replyingTo": "返信先: <1>{{name}}", "cancel": "返信をキャンセル" diff --git a/app/ui_layer/browser/frontend/src/locales/ko/chat.json b/app/ui_layer/browser/frontend/src/locales/ko/chat.json index d2f242ac..7fcd2224 100644 --- a/app/ui_layer/browser/frontend/src/locales/ko/chat.json +++ b/app/ui_layer/browser/frontend/src/locales/ko/chat.json @@ -31,6 +31,26 @@ "stopRun": "실행 중지", "stoppingRun": "실행 중지하는 중" }, + "reasoning": { + "title": "추론", + "picker": "추론 강도", + "header": "추론 · {{model}}", + "withLevel": "{{choice}} · {{level}}", + "dynamic": "모델이 결정", + "defaultTag": "기본값", + "notConfigurable": "{{model}}에서는 추론을 조정할 수 없습니다", + "clamped": "이 모델에서는 {{requested}}을(를) 사용할 수 없어 {{effective}}을(를) 사용합니다.", + "choice": { + "provider_default": "제공자 기본값", + "off": "끄기", + "minimal": "최소", + "low": "낮음", + "medium": "보통", + "high": "높음", + "xhigh": "매우 높음", + "max": "최대" + } + }, "reply": { "replyingTo": "답장 대상: <1>{{name}}", "cancel": "답장 취소" diff --git a/app/ui_layer/browser/frontend/src/locales/zh-CN/chat.json b/app/ui_layer/browser/frontend/src/locales/zh-CN/chat.json index df7a5b04..286c7ea4 100644 --- a/app/ui_layer/browser/frontend/src/locales/zh-CN/chat.json +++ b/app/ui_layer/browser/frontend/src/locales/zh-CN/chat.json @@ -31,6 +31,26 @@ "stopRun": "停止运行", "stoppingRun": "正在停止运行" }, + "reasoning": { + "title": "推理", + "picker": "推理强度", + "header": "推理 · {{model}}", + "withLevel": "{{choice}} · {{level}}", + "dynamic": "由模型决定", + "defaultTag": "默认", + "notConfigurable": "{{model}} 不支持调整推理", + "clamped": "此模型不支持“{{requested}}”,因此使用“{{effective}}”。", + "choice": { + "provider_default": "服务商默认", + "off": "关闭", + "minimal": "最低", + "low": "低", + "medium": "中", + "high": "高", + "xhigh": "超高", + "max": "最高" + } + }, "reply": { "replyingTo": "正在回复:<1>{{name}}", "cancel": "取消回复" diff --git a/app/ui_layer/browser/frontend/src/locales/zh-TW/chat.json b/app/ui_layer/browser/frontend/src/locales/zh-TW/chat.json index 7a006d1c..fe46e78d 100644 --- a/app/ui_layer/browser/frontend/src/locales/zh-TW/chat.json +++ b/app/ui_layer/browser/frontend/src/locales/zh-TW/chat.json @@ -31,6 +31,26 @@ "stopRun": "停止執行", "stoppingRun": "正在停止執行" }, + "reasoning": { + "title": "推理", + "picker": "推理強度", + "header": "推理 · {{model}}", + "withLevel": "{{choice}} · {{level}}", + "dynamic": "由模型決定", + "defaultTag": "預設", + "notConfigurable": "{{model}} 不支援調整推理", + "clamped": "此模型不支援「{{requested}}」,因此使用「{{effective}}」。", + "choice": { + "provider_default": "供應商預設", + "off": "關閉", + "minimal": "最低", + "low": "低", + "medium": "中", + "high": "高", + "xhigh": "超高", + "max": "最高" + } + }, "reply": { "replyingTo": "正在回覆:<1>{{name}}", "cancel": "取消回覆" diff --git a/app/ui_layer/browser/frontend/src/store/index.ts b/app/ui_layer/browser/frontend/src/store/index.ts index 1da895c7..9927fe6d 100644 --- a/app/ui_layer/browser/frontend/src/store/index.ts +++ b/app/ui_layer/browser/frontend/src/store/index.ts @@ -17,6 +17,7 @@ import proactiveSettingsReducer from './slices/proactiveSettingsSlice' import agentAppSettingsReducer from './slices/agentAppSettingsSlice' import generalSettingsReducer from './slices/generalSettingsSlice' import modelSettingsReducer from './slices/modelSettingsSlice' +import reasoningReducer from './slices/reasoningSlice' import integrationsSettingsReducer from './slices/integrationsSettingsSlice' import chatInputReducer from './slices/chatInputSlice' import playbooksReducer from './slices/playbooksSlice' @@ -58,6 +59,7 @@ export const store = configureStore({ agentAppSettings: agentAppSettingsReducer, generalSettings: generalSettingsReducer, modelSettings: modelSettingsReducer, + reasoning: reasoningReducer, integrationsSettings: integrationsSettingsReducer, chatInput: chatInputReducer, playbooks: playbooksReducer, diff --git a/app/ui_layer/browser/frontend/src/store/resources/catalog.ts b/app/ui_layer/browser/frontend/src/store/resources/catalog.ts index b1333396..a822018f 100644 --- a/app/ui_layer/browser/frontend/src/store/resources/catalog.ts +++ b/app/ui_layer/browser/frontend/src/store/resources/catalog.ts @@ -151,6 +151,13 @@ export const RESOURCES = { resource: 'model_settings', request: (send) => send({ type: 'slow_mode_get' }), }, + /** Chat input reasoning picker: what the model in use offers. A model or + * sign-in change is a model_settings change, so it refetches with it. */ + reasoningOptions: { + key: 'reasoningOptions', + resource: 'model_settings', + request: (send) => send({ type: 'reasoning_options_get' }), + }, // ── Static catalogs: refreshed after a reconnect ── playbooks: { diff --git a/app/ui_layer/browser/frontend/src/store/selectors/reasoning.ts b/app/ui_layer/browser/frontend/src/store/selectors/reasoning.ts new file mode 100644 index 00000000..9bf4322b --- /dev/null +++ b/app/ui_layer/browser/frontend/src/store/selectors/reasoning.ts @@ -0,0 +1,3 @@ +import type { RootState } from '../index' + +export const selectReasoningOptions = (state: RootState) => state.reasoning.options diff --git a/app/ui_layer/browser/frontend/src/store/slices/reasoningSlice.ts b/app/ui_layer/browser/frontend/src/store/slices/reasoningSlice.ts new file mode 100644 index 00000000..fe229a39 --- /dev/null +++ b/app/ui_layer/browser/frontend/src/store/slices/reasoningSlice.ts @@ -0,0 +1,52 @@ +import { createSlice, PayloadAction } from '@reduxjs/toolkit' +import type { ReasoningChoice } from '../../types' +import { register } from '../socket/messageRegistry' + +// What the chat input's reasoning picker offers for the model in use. The +// backend computes it from the live LLM interface (reasoning_options_get), +// so it tracks model switches; the CHOICE is per session and lives on the +// session (SessionInfo.reasoningEffort) or, for a draft chat, in UI state. + +export type ReasoningOptions = + /** The model has no reasoning rule: nothing is adjustable. */ + | { configurable: false; model: string } + | { + configurable: true + model: string + /** Selectable choices in display order. */ + choices: ReasoningChoice[] + /** The model's default level (marked "Default"; new chats start here). */ + defaultLevel: ReasoningChoice + /** What 'provider_default' does: 'off', a level, or 'dynamic'. */ + providerDefault: string + /** Effective choice for every stored choice (pi-style clamping). */ + resolution: Record + } + +interface ReasoningState { + options: ReasoningOptions | null +} + +const initialState: ReasoningState = { + options: null, +} + +const reasoningSlice = createSlice({ + name: 'reasoning', + initialState, + reducers: { + setReasoningOptions(state, action: PayloadAction) { + state.options = action.payload + }, + }, +}) + +const { setReasoningOptions } = reasoningSlice.actions +export default reasoningSlice.reducer + +register('reasoning_options_get', (data, dispatch) => { + const d = data as { success: false } | ({ success: true } & ReasoningOptions) + if (!d.success) return + const { success: _, ...options } = d + dispatch(setReasoningOptions(options)) +}) diff --git a/app/ui_layer/browser/frontend/src/store/uiState/catalog.ts b/app/ui_layer/browser/frontend/src/store/uiState/catalog.ts index 46c85793..096ca650 100644 --- a/app/ui_layer/browser/frontend/src/store/uiState/catalog.ts +++ b/app/ui_layer/browser/frontend/src/store/uiState/catalog.ts @@ -1,4 +1,5 @@ -import type { MetricsTimePeriod } from '../../types' +import type { MetricsTimePeriod, ReasoningChoice } from '../../types' +import { REASONING_CHOICES } from '../../types' import type { DashboardLayoutsStorage } from '../../pages/Dashboard/layout/types' import type { SettingsCategory } from '../../pages/Settings/types' import { defineUiState, defineUiStateFamily, oneOf } from './defineUiState' @@ -61,6 +62,12 @@ export const UI_STATE = { chat: { /** Speech-recognition language code; '' follows the browser language. */ micLanguage: defineUiState('chat.micLanguage', '', 'preference'), + /** The draft chat's reasoning choice, carried into the session its first + * message creates; null (untouched: the model's default level) again once + * that session exists. */ + draftReasoningEffort: defineUiState('chat.draftReasoningEffort', null, 'session', { + isValid: oneOf([null, ...REASONING_CHOICES]), + }), /** Sent inputs for ↑/↓ recall, oldest first, shared by all sessions. */ inputHistory: defineUiState('chat.inputHistory', [], 'session'), /** By session id: ids of the expanded "Action steps" chunks. */ diff --git a/app/ui_layer/browser/frontend/src/types/index.ts b/app/ui_layer/browser/frontend/src/types/index.ts index a11de1b6..333946cd 100644 --- a/app/ui_layer/browser/frontend/src/types/index.ts +++ b/app/ui_layer/browser/frontend/src/types/index.ts @@ -51,6 +51,22 @@ export const QUESTION_DISMISSED = '__dismissed__' export type SessionType = 'main' | 'chat' | 'agent_app' +/** How hard the model reasons, picked per chat session in the chat input + * (agent_core/core/models/reasoning.py ReasoningChoice). */ +export type ReasoningChoice = + | 'provider_default' + | 'off' + | 'minimal' + | 'low' + | 'medium' + | 'high' + | 'xhigh' + | 'max' + +export const REASONING_CHOICES: readonly ReasoningChoice[] = [ + 'provider_default', 'off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max', +] + export interface SessionInfo { id: string type: SessionType @@ -58,6 +74,9 @@ export interface SessionInfo { createdAt: string lastActiveAt: string agentAppProjectId?: string | null + /** Always set by the backend; null only on the client-side placeholder for + * the main session before the backend has sent it. */ + reasoningEffort: ReasoningChoice | null } // ───────────────────────────────────────────────────────────────────── diff --git a/app/ui_layer/controller/ui_controller.py b/app/ui_layer/controller/ui_controller.py index 0bd0feb2..3c500a49 100644 --- a/app/ui_layer/controller/ui_controller.py +++ b/app/ui_layer/controller/ui_controller.py @@ -6,6 +6,7 @@ from dataclasses import dataclass from typing import TYPE_CHECKING, Optional +from agent_core.core.models.reasoning import ReasoningChoice from agent_core.utils.logger import logger from app.ui_layer.events.event_bus import EventBus from app.ui_layer.events.event_types import UIEvent, UIEventType @@ -428,8 +429,17 @@ async def handle_option_click(self, value: str, session_id: str) -> None: elif value == "abort_limit": await self._agent.handle_limit_abort(session_id) - async def handle_prompt_enhance(self, user_message: str) -> str: - return await self._agent._handle_prompt_enhance(user_message=user_message) + async def handle_prompt_enhance( + self, + user_message: str, + session_id: Optional[str], + reasoning_choice: Optional[ReasoningChoice], + ) -> str: + return await self._agent._handle_prompt_enhance( + user_message=user_message, + session_id=session_id, + reasoning_choice=reasoning_choice, + ) # ───────────────────────────────────────────────────────────────────── # Event Processing diff --git a/scripts/probe_reasoning.py b/scripts/probe_reasoning.py index 6588f174..ee1dbb2c 100644 --- a/scripts/probe_reasoning.py +++ b/scripts/probe_reasoning.py @@ -1,11 +1,12 @@ # -*- coding: utf-8 -*- -"""Live check that per-model reasoning defaults are accepted by the providers. +"""Live check that per-model reasoning settings are accepted by the providers. The unit and golden tests prove what CraftBot SENDS; only the provider can say whether it ACCEPTS it. This script sends one tiny JSON request per model through the real LLMInterface (same transports, same output cap as the app) -and reports, per model, the reasoning default that was applied and whether -the provider answered or rejected it. +from inside a probe chat session holding the chosen reasoning choice (the +same session binding a real chat uses), and reports, per model, what was +sent and whether the provider answered or rejected it. Every probe is a real, billed API call (a few hundred tokens each, more for models that think). Nothing is sent with --dry-run. @@ -14,6 +15,7 @@ python scripts/probe_reasoning.py # the configured LLM model python scripts/probe_reasoning.py --model openai/gpt-5.2 --model anthropic/claude-sonnet-4-6 python scripts/probe_reasoning.py --all-rows # every table row with credentials + python scripts/probe_reasoning.py --all-rows --choice off # a session choice python scripts/probe_reasoning.py --all-rows --dry-run Exit status is 1 when any probe was rejected, else 0. @@ -44,6 +46,9 @@ #: Table surface -> the provider whose interface serves it. SUBSCRIPTION_SURFACE = "openai_subscription" +#: Id of the in-memory chat session the probes run in (never persisted). +PROBE_SESSION_ID = "reasoning-probe" + @dataclass class ProbeResult: @@ -174,6 +179,10 @@ def _probe( def main(argv: Optional[List[str]] = None) -> int: + from agent_core.core.models.reasoning import ReasoningChoice + from agent_core.core.session.session import Session + from agent_core.core.state.session import StateSession + parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) parser.add_argument( "--model", @@ -189,10 +198,22 @@ def main(argv: Optional[List[str]] = None) -> int: action="store_true", help="probe every reasoning-table row whose provider has credentials", ) + parser.add_argument( + "--choice", + type=ReasoningChoice, + default=None, + choices=list(ReasoningChoice), + metavar="CHOICE", + help=( + "the probe session's reasoning choice " + f"({', '.join(choice.value for choice in ReasoningChoice)}); " + "default: each model's default level" + ), + ) parser.add_argument( "--dry-run", action="store_true", - help="print the reasoning default per model without calling any API", + help="print what each model would be sent without calling any API", ) args = parser.parse_args(argv) @@ -205,16 +226,27 @@ def main(argv: Optional[List[str]] = None) -> int: plan = [(None, provider, model)] results: List[ProbeResult] = [] - for surface, provider, model in plan: - result = _probe(provider, model, surface, args.dry_run) - if result is None: - continue - results.append(result) - print( - f"{result.status:9s} {result.target:55s} [{result.route}] " - f"{result.decision} :: {result.detail}", - flush=True, - ) + StateSession.start( + PROBE_SESSION_ID, + current_session=Session( + id=PROBE_SESSION_ID, + reasoning_effort=args.choice.value if args.choice else None, + ), + ) + try: + with StateSession.bind(PROBE_SESSION_ID): + for surface, provider, model in plan: + result = _probe(provider, model, surface, args.dry_run) + if result is None: + continue + results.append(result) + print( + f"{result.status:9s} {result.target:55s} [{result.route}] " + f"{result.decision} :: {result.detail}", + flush=True, + ) + finally: + StateSession.end(PROBE_SESSION_ID) rejected = [r for r in results if r.status == "REJECTED"] ok = sum(r.status == "OK" for r in results) diff --git a/tests/llm/test_reasoning_rules.py b/tests/llm/test_reasoning_rules.py index 8a45f976..a375422c 100644 --- a/tests/llm/test_reasoning_rules.py +++ b/tests/llm/test_reasoning_rules.py @@ -1,44 +1,64 @@ # -*- coding: utf-8 -*- -"""Per-model reasoning defaults (agent_core/core/models/reasoning.py). +"""Per-model reasoning, chosen per chat session (agent_core/core/models/reasoning.py). -Four layers: +Five layers: 1. The table: every row is well formed, uses only the levels its wire can - carry, and is filed under a provider whose transport can send it. -2. The level policy: one level below the strongest, never above "high", - and never below the provider's own default (then the strongest level). + carry, and renders every possible choice through its provider's + transport without error. +2. The choice policy: the default level (one level below the strongest, + never above "high", never below the provider's own default), where new + sessions start, pi-style clamping of a + choice the model lacks, provider default, off, and the temperature and + output-cap consequences of each. 3. Wire rendering: each decision becomes exactly its provider's fields. -4. The transports at CraftBot's real settings (max_tokens 8000): a model with - a rule gets its fields, output cap, and temperature handling; a model - without one is untouched (the golden unruled_* snapshots additionally pin - those payloads byte for byte). +4. The transports at CraftBot's real settings (max_tokens 8000), driven by a + bound session's choice or a per-call choice; a model without a rule is + untouched (the golden unruled_* snapshots additionally pin those payloads + byte for byte). +5. Session plumbing: the stored choice, the session binding through sync + actions, and the picker's WebSocket handlers. """ from __future__ import annotations +import asyncio +from pathlib import Path +from types import SimpleNamespace +from typing import Any, Dict, List + import pytest +from agent_core.core.impl.action.executor import _atomic_action_internal_async from agent_core.core.impl.llm import reasoning_wire from agent_core.core.impl.llm.interface import LLMContextOverflowError -from agent_core.core.impl.llm.reasoning_wire import TRANSPORT_WIRES +from agent_core.core.impl.session.manager import SessionManager from agent_core.core.llm.google_gemini_client import _thinking_config from agent_core.core.models.chatgpt_subscription_client import _translate_request from agent_core.core.models.reasoning import ( BUDGET_RUNGS, BUDGET_WIRES, - KNOWN_LEVELS, + CHOICE_LADDER, LEVEL_CEILING, REASONING_RULES, + ReasoningChoice, ReasoningDecision, ReasoningRule, ReasoningWire, + clamp_choice, + reasoning_options, resolve_reasoning, - target_level, + default_choice, + default_level, ) from agent_core.core.models.registry import get_registry +from agent_core.core.session.session import Session +from agent_core.core.state.session import StateSession from .golden.conftest import GOLDEN_SYSTEM_PROMPT, build_interface +C = ReasoningChoice + #: The Anthropic SDK refuses non-streaming requests above this max_tokens. ANTHROPIC_NON_STREAMING_CEILING = 21_333 #: Anthropic and OpenRouter reject thinking budgets below this. @@ -55,6 +75,17 @@ def _auth_mode(surface: str) -> str: return "subscription" if surface == "openai_subscription" else "api_key" +def _resolve_default(provider, model, auth_mode="api_key"): + """The decision for a session holding the model's default choice.""" + return resolve_reasoning( + provider, model, auth_mode, default_choice(provider, model, auth_mode) + ) + + +#: Every reasoning level, weakest first (the ladder without "off"). +LEVELS = tuple(choice.value for choice in CHOICE_LADDER[1:]) + + ROWS = [ (surface, model, rule) for surface, rows in REASONING_RULES.items() @@ -62,6 +93,13 @@ def _auth_mode(surface: str) -> str: ] ROW_IDS = [f"{surface}/{model}" for surface, model, _ in ROWS] +_RENDERERS = { + "chat_completions": reasoning_wire.chat_completions_fields, + "anthropic_messages": reasoning_wire.anthropic_fields, + "bedrock_converse": reasoning_wire.anthropic_fields, + "gemini_native": reasoning_wire.gemini_thinking_kwargs, +} + # ─────────────────────────────── 1. the table ─────────────────────────────── @@ -71,31 +109,41 @@ def test_row_is_well_formed(surface, model, rule): assert model == model.strip() and model assert rule.levels, "a rule needs at least one level" assert len(set(rule.levels)) == len(rule.levels) - assert all(level in KNOWN_LEVELS for level in rule.levels) - assert list(rule.levels) == sorted(rule.levels, key=KNOWN_LEVELS.index) + assert all(level in LEVELS for level in rule.levels) + assert list(rule.levels) == sorted(rule.levels, key=LEVELS.index) assert rule.provider_default is None or rule.provider_default in rule.levels assert rule.source.startswith("https://") - assert rule.output_tokens >= 0 + assert rule.output_tokens >= 0 and rule.extended_output_tokens >= 0 + if rule.requires_level: + assert rule.provider_default is not None, "a required level needs a default" if rule.wire in BUDGET_WIRES: assert rule.levels == BUDGET_RUNGS assert len(rule.budgets) == len(rule.levels) assert list(rule.budgets) == sorted(set(rule.budgets)) assert rule.budgets[0] >= MIN_THINKING_BUDGET + for level in rule.levels: + # Every rung leaves room for an answer under its own cap. + assert rule.budget_for(level) < rule.output_tokens_for(level) else: assert rule.budgets == () @pytest.mark.parametrize("surface, model, rule", ROWS, ids=ROW_IDS) -def test_row_is_sendable_by_its_providers_transport(surface, model, rule): - profile = get_registry().get(_provider(surface)) - assert profile is not None, f"{surface!r} is not a registered provider" - assert rule.wire in TRANSPORT_WIRES[profile.wire] +def test_every_choice_renders_on_every_row(surface, model, rule): + render = _RENDERERS[get_registry().get(_provider(surface)).wire] + for choice in ReasoningChoice: + decision = resolve_reasoning( + _provider(surface), model, _auth_mode(surface), choice + ) + assert decision.requested is choice + render(decision) # raises if the row asks for something unsendable @pytest.mark.parametrize("surface, model, rule", ROWS, ids=ROW_IDS) def test_anthropic_caps_stay_below_the_sdk_non_streaming_ceiling(surface, model, rule): if rule.wire in (ReasoningWire.ANTHROPIC_ADAPTIVE, ReasoningWire.ANTHROPIC_BUDGET): - assert 0 < rule.output_tokens < ANTHROPIC_NON_STREAMING_CEILING + for level in rule.levels: + assert 0 < rule.output_tokens_for(level) < ANTHROPIC_NON_STREAMING_CEILING def test_duplicate_model_ids_are_refused(): @@ -124,34 +172,37 @@ def test_duplicate_model_ids_are_refused(): (("low", "high", "max"), "high", "high"), (("high",), None, "high"), (BUDGET_RUNGS, None, "high"), + # minimal is selectable but never the default, nor counted for it. + (("minimal", "low", "medium", "high"), "medium", "medium"), + (("minimal", "low", "medium", "high"), "minimal", "medium"), ], ) -def test_target_level_policy(levels, provider_default, expected): +def test_default_level_policy(levels, provider_default, expected): rule = ReasoningRule( wire=ReasoningWire.EFFORT, levels=levels, provider_default=provider_default, source="https://example.test", ) - assert target_level(rule) == expected + assert default_level(rule) == expected @pytest.mark.parametrize("surface, model, rule", ROWS, ids=ROW_IDS) -def test_every_row_resolves_within_the_policy(surface, model, rule): - decision = resolve_reasoning(_provider(surface), model, _auth_mode(surface)) +def test_default_resolves_within_the_policy_on_every_row(surface, model, rule): + decision = _resolve_default(_provider(surface), model, _auth_mode(surface)) assert decision is not None assert decision.key == f"{surface}/{model}" - chosen = rule.levels.index(decision.level) - if rule.provider_default is not None: + assert decision.choice is decision.requested and not decision.off + assert decision.requested is C(decision.level) + ladder = [level for level in rule.levels if level != "minimal"] + chosen = ladder.index(decision.level) + if rule.provider_default in ladder: # Never weaker than what the provider applies when it is omitted. - assert chosen >= rule.levels.index(rule.provider_default) - if LEVEL_CEILING in rule.levels and chosen > rule.levels.index(LEVEL_CEILING): + assert chosen >= ladder.index(rule.provider_default) + if LEVEL_CEILING in ladder and chosen > ladder.index(LEVEL_CEILING): # Above the ceiling only when the provider's own default already is. - assert rule.provider_default is not None - assert rule.levels.index(rule.provider_default) > rule.levels.index( - LEVEL_CEILING - ) - assert chosen == len(rule.levels) - 1 + assert ladder.index(rule.provider_default) > ladder.index(LEVEL_CEILING) + assert chosen == len(ladder) - 1 if decision.budget_tokens is not None: assert decision.budget_tokens < decision.output_tokens @@ -183,8 +234,8 @@ def test_every_row_resolves_within_the_policy(surface, model, rule): ("cerebras", "gpt-oss-120b", "api_key", "medium"), ], ) -def test_resolved_levels(provider, model, auth_mode, expected): - assert resolve_reasoning(provider, model, auth_mode).level == expected +def test_default_levels(provider, model, auth_mode, expected): + assert _resolve_default(provider, model, auth_mode).level == expected @pytest.mark.parametrize( @@ -197,18 +248,105 @@ def test_resolved_levels(provider, model, auth_mode, expected): ("gemini", "gemini-2.5-flash", 18_432), ], ) -def test_budget_models_resolve_to_the_high_rung(provider, model, budget): - decision = resolve_reasoning(provider, model) +def test_budget_models_default_to_the_high_rung(provider, model, budget): + decision = _resolve_default(provider, model) assert decision.level == "high" assert decision.budget_tokens == budget +@pytest.mark.parametrize( + "provider, model, requested, effective", + [ + # pi's clampThinkingLevel: nearest available at or above, else below. + ("openai", "gpt-5.2", C.MAX, C.XHIGH), + ("openai", "gpt-5.2", C.MINIMAL, C.LOW), + ("openai", "gpt-5.2", C.OFF, C.OFF), + ("openai", "gpt-5", C.OFF, C.MINIMAL), + ("openai", "gpt-5", C.XHIGH, C.HIGH), + ("openai", "gpt-6.1-sol", C.OFF, C.LOW), + ("anthropic", "claude-sonnet-4-5", C.XHIGH, C.MAX), + ("anthropic", "claude-opus-5-5", C.OFF, C.LOW), + ("gemini", "gemini-3.1-pro-preview", C.MINIMAL, C.LOW), + ("gemini", "gemini-2.5-pro", C.OFF, C.LOW), + ("glm", "glm-5.2", C.LOW, C.HIGH), + ("grok", "grok-4.5", C.OFF, C.LOW), + ("grok", "grok-4.5", C.XHIGH, C.HIGH), + # Meta-choices are never clamped, except provider_default on a model + # whose default is no reasoning, where it IS off. + ("openai", "gpt-5.5", C.PROVIDER_DEFAULT, C.PROVIDER_DEFAULT), + ("openai", "gpt-5.2", C.PROVIDER_DEFAULT, C.OFF), + ], +) +def test_choices_clamp_like_pi(provider, model, requested, effective): + decision = resolve_reasoning(provider, model, "api_key", requested) + assert decision.requested is requested + assert decision.choice is effective + assert clamp_choice(REASONING_RULES[provider][model], requested) is effective + + +def test_off_sends_nothing_where_the_default_is_no_reasoning(): + decision = resolve_reasoning("openai", "gpt-5.2", "api_key", C.OFF) + assert decision.level is None and not decision.off + assert decision.output_tokens == 0 + + +def test_off_sends_the_disable_form_where_the_model_thinks_by_default(): + decision = resolve_reasoning("openai", "gpt-5.5", "api_key", C.OFF) + assert decision.off and decision.level is None + assert decision.output_tokens == 0 + assert not decision.omit_temperature # gpt-5.5 accepts temperature at "none" + + +def test_provider_default_on_a_model_that_thinks_by_default(): + decision = resolve_reasoning("openai", "gpt-5.5", "api_key", C.PROVIDER_DEFAULT) + assert decision.level is None and not decision.off + # It reasons (at the provider's medium), so it needs the room and the + # temperature rule of a reasoning request. + assert decision.output_tokens == 32_000 + assert decision.omit_temperature + + +def test_provider_default_sends_the_documented_level_where_one_is_required(): + decision = resolve_reasoning( + "openai", "gpt-6.1-sol", "subscription", C.PROVIDER_DEFAULT + ) + assert decision.level == "low" + + +def test_claude_47_rejects_temperature_even_without_thinking(): + decision = resolve_reasoning("anthropic", "claude-opus-4-7", "api_key", C.OFF) + assert decision.level is None and not decision.off + assert decision.omit_temperature + + +def test_claude_46_keeps_temperature_when_not_thinking(): + decision = resolve_reasoning("anthropic", "claude-sonnet-4-6", "api_key", C.OFF) + assert not decision.omit_temperature + + +def test_extended_levels_get_the_extended_cap(): + assert ( + resolve_reasoning("openai", "gpt-5.2", "api_key", C.XHIGH).output_tokens + == 64_000 + ) + assert ( + resolve_reasoning("openai", "gpt-5.2", "api_key", C.HIGH).output_tokens + == 32_000 + ) + # Anthropic stays under the SDK's non-streaming ceiling at every level. + assert ( + resolve_reasoning( + "anthropic", "claude-opus-4-8", "api_key", C.MAX + ).output_tokens + == 21_000 + ) + + @pytest.mark.parametrize( "provider, model, auth_mode", [ ("openai", "gpt-4o", "api_key"), ("openai", "gpt-5.2", "subscription"), # not in the Codex catalogue - ("openai", "gpt-5.5", "api_key_typo"), # unknown auth mode: api surface ("openai", "GPT-5.2", "api_key"), # ids are exact ("openai", " gpt-5.2", "api_key"), ("openai", "gpt-5.2-pro", "api_key"), @@ -222,15 +360,30 @@ def test_budget_models_resolve_to_the_high_rung(provider, model, budget): ], ) def test_models_without_a_rule_resolve_to_none(provider, model, auth_mode): - if auth_mode == "api_key_typo": - # A non-subscription auth mode reads the public-API surface. - assert resolve_reasoning(provider, model, auth_mode).key == "openai/gpt-5.5" - return - assert resolve_reasoning(provider, model, auth_mode) is None + for choice in ReasoningChoice: + assert resolve_reasoning(provider, model, auth_mode, choice) is None + assert reasoning_options(provider, model, auth_mode) is None + + +def test_options_offer_what_the_model_accepts(): + gpt52 = reasoning_options("openai", "gpt-5.2-2025-12-11", "api_key") + # Omitting the parameter is off on gpt-5.2: no separate provider default. + assert gpt52.choices == (C.OFF, C.LOW, C.MEDIUM, C.HIGH, C.XHIGH) + assert gpt52.default_level == "high" + assert gpt52.provider_default == "off" + assert set(gpt52.resolution) == set(ReasoningChoice) + + gpt55 = reasoning_options("openai", "gpt-5.5", "api_key") + assert gpt55.choices[:2] == (C.PROVIDER_DEFAULT, C.OFF) + assert gpt55.provider_default == "medium" + + pro = reasoning_options("gemini", "gemini-2.5-pro", "api_key") + assert pro.choices == (C.PROVIDER_DEFAULT, C.LOW, C.MEDIUM, C.HIGH, C.MAX) + assert pro.provider_default == "dynamic" def test_output_cap_never_lowers_the_callers_cap(): - decision = resolve_reasoning("openai", "gpt-5.2") + decision = _resolve_default("openai", "gpt-5.2") assert decision.output_cap(APP_MAX_TOKENS) == 32_000 assert decision.output_cap(50_000) == 50_000 @@ -238,59 +391,97 @@ def test_output_cap_never_lowers_the_callers_cap(): # ─────────────────────────────── 3. rendering ─────────────────────────────── -def _decision(wire, level="high", budget=None): +def _decision(wire, level="high", budget=None, off=False): return ReasoningDecision( key="test/model", + requested=C.HIGH, + choice=C.OFF if off else C.HIGH, wire=wire, level=level, budget_tokens=budget, + off=off, output_tokens=0, omit_temperature=False, ) def test_chat_completions_fields(): - assert reasoning_wire.chat_completions_fields(_decision(ReasoningWire.EFFORT)) == ( - {"reasoning_effort": "high"}, + render = reasoning_wire.chat_completions_fields + assert render(_decision(ReasoningWire.EFFORT)) == ({"reasoning_effort": "high"}, {}) + assert render(_decision(ReasoningWire.EFFORT, level=None, off=True)) == ( + {"reasoning_effort": "none"}, + {}, + ) + assert render(_decision(ReasoningWire.EFFORT, level=None)) == ({}, {}) + assert render(_decision(ReasoningWire.OPENROUTER_EFFORT)) == ( + {}, + {"reasoning": {"effort": "high"}}, + ) + assert render(_decision(ReasoningWire.OPENROUTER_EFFORT, level=None, off=True)) == ( + {}, + {"reasoning": {"effort": "none"}}, + ) + assert render(_decision(ReasoningWire.OPENROUTER_BUDGET, budget=12_288)) == ( {}, + {"reasoning": {"max_tokens": 12_288}}, ) - assert reasoning_wire.chat_completions_fields( - _decision(ReasoningWire.OPENROUTER_EFFORT) - ) == ({}, {"reasoning": {"effort": "high"}}) - assert reasoning_wire.chat_completions_fields( - _decision(ReasoningWire.OPENROUTER_BUDGET, budget=12_288) - ) == ({}, {"reasoning": {"max_tokens": 12_288}}) def test_anthropic_fields(): - assert reasoning_wire.anthropic_fields( - _decision(ReasoningWire.ANTHROPIC_ADAPTIVE) - ) == {"thinking": {"type": "adaptive"}, "output_config": {"effort": "high"}} - assert reasoning_wire.anthropic_fields( - _decision(ReasoningWire.ANTHROPIC_BUDGET, budget=12_288) - ) == {"thinking": {"type": "enabled", "budget_tokens": 12_288}} + render = reasoning_wire.anthropic_fields + assert render(_decision(ReasoningWire.ANTHROPIC_ADAPTIVE)) == { + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "high"}, + } + assert render(_decision(ReasoningWire.ANTHROPIC_BUDGET, budget=12_288)) == { + "thinking": {"type": "enabled", "budget_tokens": 12_288} + } + assert render( + _decision(ReasoningWire.ANTHROPIC_ADAPTIVE, level=None, off=True) + ) == {"thinking": {"type": "disabled"}} + assert render(_decision(ReasoningWire.ANTHROPIC_ADAPTIVE, level=None)) == {} def test_gemini_thinking_kwargs(): - assert reasoning_wire.gemini_thinking_kwargs( - _decision(ReasoningWire.GEMINI_LEVEL, level="medium") - ) == {"thinking_level": "medium"} - assert reasoning_wire.gemini_thinking_kwargs( - _decision(ReasoningWire.GEMINI_BUDGET, budget=24_576) - ) == {"thinking_budget": 24_576} + render = reasoning_wire.gemini_thinking_kwargs + assert render(_decision(ReasoningWire.GEMINI_LEVEL, level="medium")) == { + "thinking_level": "medium" + } + assert render(_decision(ReasoningWire.GEMINI_BUDGET, budget=24_576)) == { + "thinking_budget": 24_576 + } + assert render(_decision(ReasoningWire.GEMINI_BUDGET, level=None, off=True)) == { + "thinking_budget": 0 + } + assert render(_decision(ReasoningWire.GEMINI_LEVEL, level=None)) == {} @pytest.mark.parametrize( - "render, wire", + "render, decision", [ - (reasoning_wire.chat_completions_fields, ReasoningWire.ANTHROPIC_ADAPTIVE), - (reasoning_wire.anthropic_fields, ReasoningWire.EFFORT), - (reasoning_wire.gemini_thinking_kwargs, ReasoningWire.OPENROUTER_EFFORT), + ( + reasoning_wire.chat_completions_fields, + _decision(ReasoningWire.ANTHROPIC_ADAPTIVE), + ), + (reasoning_wire.anthropic_fields, _decision(ReasoningWire.EFFORT)), + ( + reasoning_wire.gemini_thinking_kwargs, + _decision(ReasoningWire.OPENROUTER_EFFORT), + ), + # Wires without a disable form. + ( + reasoning_wire.gemini_thinking_kwargs, + _decision(ReasoningWire.GEMINI_LEVEL, level=None, off=True), + ), + ( + reasoning_wire.chat_completions_fields, + _decision(ReasoningWire.OPENROUTER_BUDGET, level=None, off=True), + ), ], ) -def test_a_wire_the_transport_cannot_send_raises(render, wire): - with pytest.raises(ValueError, match="wrong provider"): - render(_decision(wire)) +def test_what_a_transport_cannot_send_raises(render, decision): + with pytest.raises(ValueError, match="cannot send"): + render(decision) def test_gemini_thinking_config_builder(): @@ -315,6 +506,24 @@ def test_codex_translator_uses_the_resolved_effort(): # ────────────────────── 4. transports at CraftBot's settings ───────────────── +@pytest.fixture() +def bound_session(): + """Register a session with a given reasoning choice and bind to it.""" + registered: List[str] = [] + + def _bind(choice: ReasoningChoice, session_id: str = "s-reasoning-test"): + StateSession.start( + session_id, + current_session=Session(id=session_id, reasoning_effort=choice.value), + ) + registered.append(session_id) + return StateSession.bind(session_id) + + yield _bind + for session_id in registered: + StateSession.end(session_id) + + def _one_call(monkeypatch, provider, model, auth_mode="api_key"): iface, rec = build_interface(monkeypatch, provider, model) iface.max_tokens = APP_MAX_TOKENS @@ -324,7 +533,7 @@ def _one_call(monkeypatch, provider, model, auth_mode="api_key"): return iface, rec.calls[0]["payload"] -def test_openai_ruled_model(monkeypatch): +def test_openai_default_level(monkeypatch): _, payload = _one_call(monkeypatch, "openai", "gpt-5.2-2025-12-11") assert payload["reasoning_effort"] == "high" assert payload["max_completion_tokens"] == 32_000 @@ -332,12 +541,41 @@ def test_openai_ruled_model(monkeypatch): assert payload["response_format"] == {"type": "json_object"} -def test_openai_unruled_model(monkeypatch): - _, payload = _one_call(monkeypatch, "openai", "gpt-4o") +def test_openai_unruled_model(monkeypatch, bound_session): + with bound_session(C.XHIGH): + _, payload = _one_call(monkeypatch, "openai", "gpt-4o") assert "reasoning_effort" not in payload assert payload["max_completion_tokens"] == APP_MAX_TOKENS +def test_bound_session_choice_drives_the_request(monkeypatch, bound_session): + with bound_session(C.XHIGH): + _, payload = _one_call(monkeypatch, "openai", "gpt-5.2-2025-12-11") + assert payload["reasoning_effort"] == "xhigh" + assert payload["max_completion_tokens"] == 64_000 + + +def test_bound_session_off(monkeypatch, bound_session): + with bound_session(C.OFF): + _, payload = _one_call(monkeypatch, "openai", "gpt-5.5") + assert payload["reasoning_effort"] == "none" + assert payload["max_completion_tokens"] == APP_MAX_TOKENS + + +def test_per_call_choice_wins_over_the_bound_session(monkeypatch, bound_session): + iface, rec = build_interface(monkeypatch, "openai", "gpt-5.2-2025-12-11") + iface.max_tokens = APP_MAX_TOKENS + with bound_session(C.LOW): + asyncio.run( + iface.generate_response_async( + system_prompt=GOLDEN_SYSTEM_PROMPT, + user_prompt="hi", + reasoning_choice=C.MEDIUM, + ) + ) + assert rec.calls[0]["payload"]["reasoning_effort"] == "medium" + + def test_subscription_request_carries_the_codex_row(monkeypatch): _, payload = _one_call(monkeypatch, "openai", "gpt-5.5", auth_mode="subscription") assert payload["reasoning_effort"] == "high" @@ -363,7 +601,7 @@ def test_cerebras_cap_is_clamped_by_the_profile(monkeypatch): def test_glm_bad_model_is_pinned_at_its_max(monkeypatch): _, payload = _one_call(monkeypatch, "glm", "glm-5.3") assert payload["reasoning_effort"] == "max" - assert payload["max_tokens"] == 32_000 + assert payload["max_tokens"] == 64_000 def test_openrouter_openai_row(monkeypatch): @@ -374,7 +612,7 @@ def test_openrouter_openai_row(monkeypatch): assert payload["max_tokens"] == 32_000 -def test_anthropic_ruled_model(monkeypatch): +def test_anthropic_default_level(monkeypatch): _, payload = _one_call(monkeypatch, "anthropic", "claude-opus-5-5") assert payload["thinking"] == {"type": "adaptive"} assert payload["output_config"] == {"effort": "high"} @@ -382,6 +620,23 @@ def test_anthropic_ruled_model(monkeypatch): assert "extra_body" not in payload # no temperature +def test_anthropic_47_provider_default_still_drops_temperature( + monkeypatch, bound_session +): + with bound_session(C.PROVIDER_DEFAULT): + _, payload = _one_call(monkeypatch, "anthropic", "claude-opus-4-7") + assert "thinking" not in payload and "output_config" not in payload + assert "extra_body" not in payload + assert payload["max_tokens"] == 16_384 + + +def test_anthropic_46_off_keeps_temperature(monkeypatch, bound_session): + with bound_session(C.OFF): + _, payload = _one_call(monkeypatch, "anthropic", "claude-sonnet-4-6") + assert "thinking" not in payload + assert payload["extra_body"] == {"temperature": 0.0} + + def test_anthropic_unruled_model(monkeypatch): _, payload = _one_call(monkeypatch, "anthropic", "claude-3-5-haiku-20241022") assert "thinking" not in payload and "output_config" not in payload @@ -398,6 +653,13 @@ def test_bedrock_adaptive_row(monkeypatch): assert payload["inferenceConfig"] == {"maxTokens": 21_000} +def test_bedrock_off_disables_thinking(monkeypatch, bound_session): + with bound_session(C.OFF): + _, payload = _one_call(monkeypatch, "bedrock", "us.anthropic.claude-opus-5") + assert payload["additionalModelRequestFields"] == {"thinking": {"type": "disabled"}} + assert payload["inferenceConfig"] == {"maxTokens": APP_MAX_TOKENS} + + def test_bedrock_unruled_model(monkeypatch): _, payload = _one_call(monkeypatch, "bedrock", "meta.llama3-3-70b-instruct-v1:0") assert "additionalModelRequestFields" not in payload @@ -414,17 +676,35 @@ def test_gemini_level_row(monkeypatch): assert config["maxOutputTokens"] == 32_768 -def test_gemini_caller_budget_wins_over_the_default(monkeypatch): +def test_gemini_off_is_a_zero_budget(monkeypatch, bound_session): + with bound_session(C.OFF): + _, payload = _one_call(monkeypatch, "gemini", "gemini-2.5-flash") + config = payload["body"]["generationConfig"] + assert config["thinkingConfig"] == {"thinkingBudget": 0} + assert config["maxOutputTokens"] == APP_MAX_TOKENS + + +def test_gemini_provider_default_sends_no_thinking_config(monkeypatch, bound_session): + with bound_session(C.PROVIDER_DEFAULT): + _, payload = _one_call(monkeypatch, "gemini", "gemini-2.5-pro") + config = payload["body"]["generationConfig"] + assert "thinkingConfig" not in config + # 2.5 Pro still thinks dynamically, so the request keeps the room. + assert config["maxOutputTokens"] == 32_768 + + +def test_gemini_caller_budget_wins_over_the_session(monkeypatch, bound_session): iface, rec = build_interface(monkeypatch, "gemini", "gemini-3.5-flash") iface.max_tokens = APP_MAX_TOKENS iface._begin_call(thinking_budget=512) - iface._generate_response_sync(GOLDEN_SYSTEM_PROMPT, "hi") + with bound_session(C.HIGH): + iface._generate_response_sync(GOLDEN_SYSTEM_PROMPT, "hi") config = rec.calls[0]["payload"]["body"]["generationConfig"] assert config["thinkingConfig"] == {"thinkingBudget": 512} assert config["maxOutputTokens"] == APP_MAX_TOKENS -def test_context_check_reserves_the_reasoning_cap(monkeypatch): +def test_context_check_reserves_the_reasoning_cap(monkeypatch, bound_session): import app.config as app_config monkeypatch.setattr(app_config, "get_context_window", lambda: 40_000) @@ -441,9 +721,201 @@ def test_context_check_reserves_the_reasoning_cap(monkeypatch): with pytest.raises(LLMContextOverflowError, match="32000 reserved for output"): ruled._check_context_fits(None, prompt) + # A session that turned reasoning off reserves only the caller's cap. + with bound_session(C.OFF): + assert ruled._output_reservation() == APP_MAX_TOKENS + ruled._check_context_fits(None, prompt) + def test_decision_follows_a_model_switch(monkeypatch): iface, _ = build_interface(monkeypatch, "openai", "gpt-4o") assert iface.reasoning_decision() is None + assert iface.reasoning_options() is None iface.model = "gpt-5.2" assert iface.reasoning_decision().level == "high" + assert iface.reasoning_options().default_level == "high" + + +# ──────────────────────────── 5. session plumbing ─────────────────────────── + + +def test_session_round_trips_its_choice(): + session = Session(id="s1", reasoning_effort=C.XHIGH.value) + assert Session.from_dict(session.to_dict()).reasoning_effort == "xhigh" + # Absent or no longer valid (the retired "auto"): discarded, so the + # session manager reseeds it on restore. + assert Session.from_dict({"id": "s2"}).reasoning_effort is None + assert ( + Session.from_dict({"id": "s3", "reasoning_effort": "auto"}).reasoning_effort + is None + ) + assert ( + Session.from_dict({"id": "s4", "reasoning_effort": "turbo"}).reasoning_effort + is None + ) + + +@pytest.mark.parametrize( + "provider, model, expected", + [ + ("openai", "gpt-5.2-2025-12-11", C.HIGH), + ("groq", "openai/gpt-oss-120b", C.MEDIUM), + ("glm", "glm-5.3", C.MAX), + # No rule: the default ceiling, clamped once a model with a rule runs. + ("openai", "gpt-4o", C.HIGH), + (None, None, C.HIGH), + ], +) +def test_default_choice_a_new_session_starts_at(provider, model, expected): + assert default_choice(provider, model) is expected + + +def test_new_and_restored_sessions_are_seeded_with_the_model_default( + monkeypatch, tmp_path: Path +): + iface, _ = build_interface(monkeypatch, "groq", "openai/gpt-oss-120b") + persisted: List[Dict[str, Any]] = [] + manager = SessionManager( + event_stream_manager=None, + llm_interface=iface, + workspace_root=tmp_path, + on_session_persist=lambda session: persisted.append(session.to_dict()), + ) + created = manager.create_session() + restored = manager.restore_session( + Session.from_dict({"id": "s-restored", "reasoning_effort": "auto"}) + ) + kept = manager.restore_session( + Session.from_dict({"id": "s-kept", "reasoning_effort": "xhigh"}) + ) + try: + assert created.reasoning_effort == "medium" + # The retired value is discarded and reseeded, and the seed is saved. + assert restored.reasoning_effort == "medium" + assert persisted[-1] == { + **persisted[-1], + "id": "s-restored", + "reasoning_effort": "medium", + } + # A valid stored choice is kept as is, even one the model lacks. + assert kept.reasoning_effort == "xhigh" + finally: + for session_id in (created.id, "s-restored", "s-kept"): + StateSession.end(session_id) + + +def test_work_outside_any_session_uses_the_model_default(monkeypatch): + iface, _ = build_interface(monkeypatch, "glm", "glm-5.3") + assert iface.reasoning_choice() is C.MAX + assert iface.reasoning_decision().level == "max" + + +def test_session_manager_creates_and_sets_the_choice(tmp_path: Path): + persisted: List[Dict[str, Any]] = [] + manager = SessionManager( + event_stream_manager=None, + workspace_root=tmp_path, + on_session_persist=lambda session: persisted.append(session.to_dict()), + ) + session = manager.create_session(reasoning_effort=C.LOW) + try: + assert session.reasoning_effort == "low" + assert manager.set_reasoning_effort(session.id, C.OFF) + assert manager.get(session.id).reasoning_effort == "off" + assert persisted[-1]["reasoning_effort"] == "off" + assert not manager.set_reasoning_effort("no-such-session", C.OFF) + finally: + StateSession.end(session.id) + + +def test_sync_actions_run_in_the_callers_session_binding(bound_session): + code = ( + "def probe(input_data):\n" + " from agent_core.core.state.session import StateSession\n" + " state = StateSession.bound()\n" + " return {'session': state.session_id if state else None}\n" + ) + + async def _run(): + with bound_session(C.HIGH, session_id="s-sync-action"): + return await _atomic_action_internal_async("probe", code, {}, "CLI") + + assert asyncio.run(_run()) == {"session": "s-sync-action"} + + +def _adapter_with(agent) -> Any: + from app.ui_layer.adapters.browser_adapter import BrowserAdapter + + adapter = BrowserAdapter.__new__(BrowserAdapter) + adapter._controller = SimpleNamespace(agent=agent) + adapter.sent = [] + adapter.broadcasts = [] + + async def _send_to(ws, message): + adapter.sent.append(message) + + async def _broadcast(message): + adapter.broadcasts.append(message) + + adapter._send_to = _send_to + adapter._broadcast = _broadcast + return adapter + + +def test_reasoning_options_handler(monkeypatch): + iface, _ = build_interface(monkeypatch, "openai", "gpt-5.2-2025-12-11") + adapter = _adapter_with(SimpleNamespace(llm=iface)) + asyncio.run(adapter._handle_reasoning_options_get(ws=None)) + data = adapter.sent[0]["data"] + assert data["configurable"] is True + assert data["choices"] == ["off", "low", "medium", "high", "xhigh"] + assert data["defaultLevel"] == "high" + assert data["providerDefault"] == "off" + assert data["resolution"]["max"] == "xhigh" + + unruled, _ = build_interface(monkeypatch, "openai", "gpt-4o") + adapter = _adapter_with(SimpleNamespace(llm=unruled)) + asyncio.run(adapter._handle_reasoning_options_get(ws=None)) + assert adapter.sent[0]["data"] == { + "success": True, + "model": "gpt-4o", + "configurable": False, + } + + +def test_session_reasoning_set_handler(tmp_path: Path): + manager = SessionManager(event_stream_manager=None, workspace_root=tmp_path) + session = manager.create_session() + agent = SimpleNamespace( + session_manager=manager, + set_session_reasoning_effort=manager.set_reasoning_effort, + ) + adapter = _adapter_with(agent) + try: + asyncio.run( + adapter._handle_session_reasoning_set( + {"sessionId": session.id, "reasoningEffort": "minimal"} + ) + ) + assert session.reasoning_effort == "minimal" + assert adapter.broadcasts[-1]["data"]["session"]["reasoningEffort"] == "minimal" + + # An unknown value changes nothing. + asyncio.run( + adapter._handle_session_reasoning_set( + {"sessionId": session.id, "reasoningEffort": "turbo"} + ) + ) + assert session.reasoning_effort == "minimal" + assert len(adapter.broadcasts) == 1 + finally: + StateSession.end(session.id) + + +def test_draft_reasoning_from_a_message(): + from app.ui_layer.adapters.browser_adapter import BrowserAdapter + + assert BrowserAdapter._draft_reasoning({"reasoningEffort": "xhigh"}) is C.XHIGH + assert BrowserAdapter._draft_reasoning({}) is None + with pytest.raises(ValueError): + BrowserAdapter._draft_reasoning({"reasoningEffort": "turbo"}) From 10ffd5de1ffb1a32a0c47835e9437915b8aed9d1 Mon Sep 17 00:00:00 2001 From: CraftBot Date: Fri, 2 Oct 2026 09:27:32 +0900 Subject: [PATCH 3/5] fix max token ignores the reasoning output cap bug --- agent_core/core/impl/llm/interface.py | 8 +++++++- tests/llm/test_reasoning_rules.py | 29 +++++++++++++++++++++++++++ 2 files changed, 36 insertions(+), 1 deletion(-) diff --git a/agent_core/core/impl/llm/interface.py b/agent_core/core/impl/llm/interface.py index 520330cf..ff75639b 100644 --- a/agent_core/core/impl/llm/interface.py +++ b/agent_core/core/impl/llm/interface.py @@ -1223,6 +1223,11 @@ def fits_context( plus the output reservation, against the window less the headroom the summary request needs. On a session's first request there is no provider count yet, so the whole prompt is counted locally. + + The output reservation is the one the pre-send check + (_check_context_fits) applies, including the reasoning output cap of + the session's choice, so the stream folds before that check could + refuse the request. """ from app.config import get_context_window, get_reserve_tokens @@ -1232,7 +1237,8 @@ def fits_context( else: projected = last + count_tokens(pending) return ( - projected + self.max_tokens <= get_context_window() - get_reserve_tokens() + projected + self._output_reservation() + <= get_context_window() - get_reserve_tokens() ) def end_all_session_caches(self, task_id: str) -> None: diff --git a/tests/llm/test_reasoning_rules.py b/tests/llm/test_reasoning_rules.py index a375422c..26448e4a 100644 --- a/tests/llm/test_reasoning_rules.py +++ b/tests/llm/test_reasoning_rules.py @@ -727,6 +727,35 @@ def test_context_check_reserves_the_reasoning_cap(monkeypatch, bound_session): ruled._check_context_fits(None, prompt) +def test_fold_check_reserves_the_same_cap_as_the_pre_send_check( + monkeypatch, bound_session +): + import app.config as app_config + + # window 128,000 - reserve 16,384 - the output reservation. + monkeypatch.setattr(app_config, "get_context_window", lambda: 128_000) + monkeypatch.setattr(app_config, "get_reserve_tokens", lambda: 16_384) + + def assert_folds_after(iface, last_input_tokens: int) -> None: + key = "task:action_selection" + iface._last_input_tokens[key] = last_input_tokens + assert iface.fits_context("task", "action_selection", "s", "") + iface._last_input_tokens[key] = last_input_tokens + 1 + assert not iface.fits_context("task", "action_selection", "s", "") + + unruled, _ = build_interface(monkeypatch, "openai", "gpt-4o") + unruled.max_tokens = APP_MAX_TOKENS + assert_folds_after(unruled, 103_616) # the caller's 8,000 + + ruled, _ = build_interface(monkeypatch, "openai", "gpt-5.2-2025-12-11") + ruled.max_tokens = APP_MAX_TOKENS + assert_folds_after(ruled, 79_616) # default high: 32,000 + with bound_session(C.XHIGH): + assert_folds_after(ruled, 47_616) # 64,000 + with bound_session(C.OFF): + assert_folds_after(ruled, 103_616) # no reasoning: the caller's 8,000 + + def test_decision_follows_a_model_switch(monkeypatch): iface, _ = build_interface(monkeypatch, "openai", "gpt-4o") assert iface.reasoning_decision() is None From 0c054e18b538b2e394a923e5cf3acb4ae604dbeb Mon Sep 17 00:00:00 2001 From: CraftBot Date: Fri, 2 Oct 2026 10:00:12 +0900 Subject: [PATCH 4/5] openai setting update --- .../models/chatgpt_subscription_client.py | 17 +++--- agent_core/core/models/factory.py | 23 +++----- agent_core/core/models/provider_config.py | 20 +++++-- agent_core/core/models/reasoning.py | 21 +++---- agent_file_system/AGENT.md | 2 +- app/data/agent_file_system_template/AGENT.md | 2 +- craftos_integrations/llm_oauth/chatgpt.py | 29 ---------- tests/integrations/test_import_surface.py | 1 - tests/llm/test_reasoning_rules.py | 56 ++++++++++++++++--- 9 files changed, 91 insertions(+), 80 deletions(-) diff --git a/agent_core/core/models/chatgpt_subscription_client.py b/agent_core/core/models/chatgpt_subscription_client.py index c2e82fef..520470b2 100644 --- a/agent_core/core/models/chatgpt_subscription_client.py +++ b/agent_core/core/models/chatgpt_subscription_client.py @@ -163,22 +163,23 @@ def __init__( # silently honoring "best-effort" semantics is fine for fields that just # don't apply at this backend (e.g. ``max_tokens`` becomes "let the # server decide" rather than a hard failure). -#: Effort for a model the reasoning table has no row for. The Codex backend -#: requires a ``reasoning`` block on every request, so unlike every other -#: provider the parameter cannot simply be left out; "medium" is the Codex -#: CLI default and what this translator always sent before per-model rules. -_CODEX_UNRULED_EFFORT = "medium" +#: Effort for a request that carries no ``reasoning_effort``. LLM requests +#: always carry one (every subscription model has a reasoning row); VLM +#: requests send none. The Codex backend requires a ``reasoning`` block on +#: every request, so unlike every other provider the parameter cannot simply +#: be left out; "medium" is the Codex CLI default. +_CODEX_DEFAULT_EFFORT = "medium" def _codex_reasoning_config(requested_effort: Any) -> Dict[str, str]: """Build the ``reasoning`` block Codex requires on every request. ``requested_effort`` is the caller's Chat-Completions - ``reasoning_effort``: the per-model default the chat_completions - transport resolved, or None when the model has no rule. + ``reasoning_effort``: the level the chat_completions transport resolved + for the session's reasoning choice, or None when the caller sends none. ``"auto"`` summary matches the Codex CLI. """ - effort = requested_effort if requested_effort else _CODEX_UNRULED_EFFORT + effort = requested_effort if requested_effort else _CODEX_DEFAULT_EFFORT return {"effort": str(effort), "summary": "auto"} diff --git a/agent_core/core/models/factory.py b/agent_core/core/models/factory.py index e360c96b..7d8162f8 100644 --- a/agent_core/core/models/factory.py +++ b/agent_core/core/models/factory.py @@ -353,25 +353,18 @@ def create( access_token, sub_base_url, extra_headers = oauth - # Codex's accepted-model list lives in the ChatGPT OAuth - # backend module so provider-specific knowledge stays - # colocated with the flow that authenticates against it. - # See ``llm_oauth.chatgpt.CODEX_ACCEPTED_MODELS`` for the - # source-of-truth list and the reasoning behind the fallback. - from craftos_integrations.llm_oauth.chatgpt import ( - CODEX_ACCEPTED_MODELS, - effective_model_for_subscription, - ) - - effective_model, was_substituted = effective_model_for_subscription( - model - ) - if was_substituted: + # The Codex backend serves only the profile's + # subscription_models; any other model runs as the + # profile's subscription default. + if model in cfg.subscription_models: + effective_model = model + else: + effective_model = cfg.subscription_default_model logger.warning( f"[FACTORY] ChatGPT subscription mode rejects model " f"{model!r}; substituting {effective_model!r}. " f"Valid Codex-subscription models: " - f"{sorted(CODEX_ACCEPTED_MODELS)}. Set the model in " + f"{list(cfg.subscription_models)}. Set the model in " f"Settings to silence this warning." ) diff --git a/agent_core/core/models/provider_config.py b/agent_core/core/models/provider_config.py index 51fdaba8..2be66cff 100644 --- a/agent_core/core/models/provider_config.py +++ b/agent_core/core/models/provider_config.py @@ -179,14 +179,24 @@ def resolve_temperature(profile: Optional[ProviderProfile], caller_temperature): settings_key="openai", oauth_backend="chatgpt", subscription_label="Sign in with ChatGPT", - # Codex-accepted models for ChatGPT subscription auth. + # The only models ChatGPT-subscription auth runs: the Codex catalogue's + # listed models, in its order + # (https://github.com/openai/codex/blob/main/codex-rs/models-manager/models.json). + # The factory runs any other model as subscription_default_model. + # Codex requires a reasoning level on every request, so each of + # these has a row in agent_core/core/models/reasoning.py + # (tests/llm/test_reasoning_rules.py checks they agree). subscription_models=( - "gpt-5.4", + "gpt-6.1-sol", + "gpt-6-astra", + "gpt-6-sol", + "gpt-6-luna", + "gpt-5.6-sol", + "gpt-5.6-terra", + "gpt-5.6-luna", "gpt-5.5", - "gpt-5.4-mini", - "gpt-5.3-codex-spark", ), - subscription_default_model="gpt-5.4", + subscription_default_model="gpt-6.1-sol", supports_prompt_cache_key=True, # OpenAI deprecated `max_tokens` in favor of `max_completion_tokens`; # every current chat model accepts the new field, and reasoning diff --git a/agent_core/core/models/reasoning.py b/agent_core/core/models/reasoning.py index 8d6d5baf..40cb97ff 100644 --- a/agent_core/core/models/reasoning.py +++ b/agent_core/core/models/reasoning.py @@ -466,12 +466,12 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: # ── OpenAI public API (Chat Completions ``reasoning_effort``) ── # Values and defaults: https://developers.openai.com/api/docs/models/. -# Models before gpt-5.1 reject "none" and never accept temperature; gpt-5.1+ -# reject "minimal", default to no reasoning through gpt-5.4, and accept -# temperature only at "none". xhigh arrived with gpt-5.2. "max" exists only on -# the Responses API: Chat Completions rejects it on gpt-5.6 and GPT-6 ("Supported -# values are: 'none', 'low', 'medium', 'high', and 'xhigh'", live API error, -# 2026-10-02; GPT-6 Astra and 6.1 Sol list the same without 'none'). +# Models before gpt-5.1 reject "none"; gpt-5.1+ reject "minimal" and default +# to no reasoning through gpt-5.4. xhigh arrived with gpt-5.2. "max" exists +# only on the Responses API: Chat Completions rejects it on gpt-5.6 and GPT-6 +# ("Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'", live +# API error, 2026-10-02; GPT-6 Astra and 6.1 Sol list the same without 'none'). +# Temperature is the OpenAI profile's: it omits it on every request. _OPENAI_DOCS = "https://developers.openai.com/api/docs/models" @@ -481,7 +481,6 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: provider_default="medium", source=f"{_OPENAI_DOCS}/gpt-5", output_tokens=_EFFORT_OUTPUT_TOKENS, - temperature=TemperaturePolicy.OMIT_ALWAYS, ) _OPENAI_O_SERIES = ReasoningRule( wire=ReasoningWire.EFFORT, @@ -489,7 +488,6 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: provider_default="medium", source="https://developers.openai.com/api/docs/guides/reasoning", output_tokens=_EFFORT_OUTPUT_TOKENS, - temperature=TemperaturePolicy.OMIT_ALWAYS, ) _OPENAI_GPT51 = ReasoningRule( wire=ReasoningWire.EFFORT, @@ -498,7 +496,6 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: source=f"{_OPENAI_DOCS}/gpt-5.1", output_tokens=_EFFORT_OUTPUT_TOKENS, off=ReasoningOff.OMIT, - temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) _OPENAI_GPT52_TO_54 = ReasoningRule( wire=ReasoningWire.EFFORT, @@ -508,7 +505,6 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: output_tokens=_EFFORT_OUTPUT_TOKENS, extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, off=ReasoningOff.OMIT, - temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) _OPENAI_GPT55 = ReasoningRule( wire=ReasoningWire.EFFORT, @@ -518,7 +514,6 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: output_tokens=_EFFORT_OUTPUT_TOKENS, extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, off=ReasoningOff.EXPLICIT, - temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) _OPENAI_GPT56_PLUS = ReasoningRule( wire=ReasoningWire.EFFORT, @@ -528,7 +523,6 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: output_tokens=_EFFORT_OUTPUT_TOKENS, extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, off=ReasoningOff.EXPLICIT, - temperature=TemperaturePolicy.OMIT_WHILE_THINKING, ) #: GPT-6 Astra and GPT-6.1 Sol reject "none" and always reason. _OPENAI_GPT6_ALWAYS_REASONING = ReasoningRule( @@ -538,7 +532,6 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: source=f"{_OPENAI_DOCS}/gpt-6-astra", output_tokens=_EFFORT_OUTPUT_TOKENS, extended_output_tokens=_EFFORT_EXTENDED_OUTPUT_TOKENS, - temperature=TemperaturePolicy.OMIT_ALWAYS, ) _OPENAI_GPT61_SOL = replace( _OPENAI_GPT6_ALWAYS_REASONING, @@ -552,6 +545,8 @@ def _bedrock_ids(base_id: str, *profile_prefixes: str) -> Tuple[str, ...]: # supported_reasoning_levels minus the client-side "ultra" alias; no model # lists "none", so none can be turned off # (https://github.com/openai/codex/blob/main/codex-rs/models-manager/models.json). +# The rows are exactly the OpenAI profile's subscription_models, the only +# models subscription auth runs. _CODEX_CATALOG = ( "https://github.com/openai/codex/blob/main/codex-rs/models-manager/models.json" diff --git a/agent_file_system/AGENT.md b/agent_file_system/AGENT.md index 7698f9e3..2036db75 100644 --- a/agent_file_system/AGENT.md +++ b/agent_file_system/AGENT.md @@ -3078,7 +3078,7 @@ Users can authenticate OpenAI or Grok by signing in to their paid subscription ( ChatGPT subscription specifics: - Requests route through OpenAI's Codex backend. CraftBot's JSON-mode action decisions work transparently; only native tool-calls (`tools=[...]`) and streaming are unsupported — neither is CraftBot's normal path, so actions run fine. -- Codex accepts a fixed model set (gpt-5.5, gpt-5.4, gpt-5.4-mini, gpt-5.3-codex-spark; default gpt-5.4); any other model name is silently substituted. +- Codex accepts a fixed model set (gpt-6.1-sol, gpt-6-astra, gpt-6-sol, gpt-6-luna, gpt-5.6-sol, gpt-5.6-terra, gpt-5.6-luna, gpt-5.5; default gpt-6.1-sol); any other model name is silently substituted. - The real hard failure is a Free-tier account (no Plus/Pro/Team): `CHATGPT_SUBSCRIPTION_REJECTED`. That's the "upgrade or switch to an API key" case — do not retry. - If the credential is disconnected mid-session, the client raises an actionable error telling the user to re-save model settings or reconnect. diff --git a/app/data/agent_file_system_template/AGENT.md b/app/data/agent_file_system_template/AGENT.md index 7698f9e3..2036db75 100644 --- a/app/data/agent_file_system_template/AGENT.md +++ b/app/data/agent_file_system_template/AGENT.md @@ -3078,7 +3078,7 @@ Users can authenticate OpenAI or Grok by signing in to their paid subscription ( ChatGPT subscription specifics: - Requests route through OpenAI's Codex backend. CraftBot's JSON-mode action decisions work transparently; only native tool-calls (`tools=[...]`) and streaming are unsupported — neither is CraftBot's normal path, so actions run fine. -- Codex accepts a fixed model set (gpt-5.5, gpt-5.4, gpt-5.4-mini, gpt-5.3-codex-spark; default gpt-5.4); any other model name is silently substituted. +- Codex accepts a fixed model set (gpt-6.1-sol, gpt-6-astra, gpt-6-sol, gpt-6-luna, gpt-5.6-sol, gpt-5.6-terra, gpt-5.6-luna, gpt-5.5; default gpt-6.1-sol); any other model name is silently substituted. - The real hard failure is a Free-tier account (no Plus/Pro/Team): `CHATGPT_SUBSCRIPTION_REJECTED`. That's the "upgrade or switch to an API key" case — do not retry. - If the credential is disconnected mid-session, the client raises an actionable error telling the user to re-save model settings or reconnect. diff --git a/craftos_integrations/llm_oauth/chatgpt.py b/craftos_integrations/llm_oauth/chatgpt.py index e509d2cc..82bf36d1 100644 --- a/craftos_integrations/llm_oauth/chatgpt.py +++ b/craftos_integrations/llm_oauth/chatgpt.py @@ -58,35 +58,6 @@ logger = get_logger(__name__) -# ════════════════════════════════════════════════════════════════════════ -# Accepted-model list for ChatGPT-subscription auth -# ════════════════════════════════════════════════════════════════════════ - -CODEX_ACCEPTED_MODELS = frozenset( - { - "gpt-5.5", - "gpt-5.4", - "gpt-5.4-mini", - "gpt-5.3-codex-spark", - } -) - -CODEX_DEFAULT_MODEL = "gpt-5.4" - - -def effective_model_for_subscription(model: str) -> Tuple[str, bool]: - """Return ``(effective, was_substituted)`` for a Codex-subscription call. - - If ``model`` is one of the accepted names it passes through - unchanged. Otherwise it's replaced with ``CODEX_DEFAULT_MODEL`` and - the second return value is ``True`` so the caller can log the - substitution once. - """ - if model in CODEX_ACCEPTED_MODELS: - return model, False - return CODEX_DEFAULT_MODEL, True - - AUTH_URL = "https://auth.openai.com/oauth/authorize" TOKEN_URL = "https://auth.openai.com/oauth/token" diff --git a/tests/integrations/test_import_surface.py b/tests/integrations/test_import_surface.py index 6088a387..2ea96f2b 100644 --- a/tests/integrations/test_import_surface.py +++ b/tests/integrations/test_import_surface.py @@ -46,7 +46,6 @@ def test_llm_oauth_entry_points(): assert callable(tokens.get_bearer) assert callable(tokens.status) assert callable(chatgpt.load) - assert chatgpt.CODEX_ACCEPTED_MODELS assert callable(grok.load) # Not connected on a clean checkout, but the call must work. diff --git a/tests/llm/test_reasoning_rules.py b/tests/llm/test_reasoning_rules.py index 26448e4a..78abecc8 100644 --- a/tests/llm/test_reasoning_rules.py +++ b/tests/llm/test_reasoning_rules.py @@ -154,6 +154,38 @@ def test_duplicate_model_ids_are_refused(): reasoning._surface({"m": rule}, {"m": rule}) +def test_subscription_rows_are_exactly_the_models_subscription_auth_runs(): + # Codex requires a level on every request, so a subscription model + # without a row would run at Codex's fallback, and a row for a model + # subscription auth never runs would be dead. + profile = get_registry().get("openai") + assert set(REASONING_RULES["openai_subscription"]) == set( + profile.subscription_models + ) + assert profile.subscription_default_model in profile.subscription_models + + +def test_subscription_runs_an_unlisted_model_as_the_default(monkeypatch): + from agent_core.core.models import factory + from agent_core.core.models.types import InterfaceType + + monkeypatch.setattr( + factory, + "_get_oauth_bearer", + lambda provider: ("token", "https://chatgpt.com/backend-api/codex", {}), + ) + for requested, runs in ( + ("gpt-6-sol", "gpt-6-sol"), + ("gpt-5.4", "gpt-6.1-sol"), + ("gpt-5.2-2025-12-11", "gpt-6.1-sol"), + ): + ctx = factory.ModelFactory.create( + provider="openai", interface=InterfaceType.LLM, model_override=requested + ) + assert ctx["auth_mode"] == "subscription" + assert ctx["model"] == runs + + # ─────────────────────────────── 2. the policy ────────────────────────────── @@ -294,16 +326,25 @@ def test_off_sends_the_disable_form_where_the_model_thinks_by_default(): decision = resolve_reasoning("openai", "gpt-5.5", "api_key", C.OFF) assert decision.off and decision.level is None assert decision.output_tokens == 0 - assert not decision.omit_temperature # gpt-5.5 accepts temperature at "none" def test_provider_default_on_a_model_that_thinks_by_default(): decision = resolve_reasoning("openai", "gpt-5.5", "api_key", C.PROVIDER_DEFAULT) assert decision.level is None and not decision.off - # It reasons (at the provider's medium), so it needs the room and the - # temperature rule of a reasoning request. + # It reasons (at the provider's medium), so it needs the room of a + # reasoning request. assert decision.output_tokens == 32_000 - assert decision.omit_temperature + + +def test_temperature_is_dropped_exactly_while_the_request_reasons(): + # OpenRouter forwards an explicit temperature upstream, where reasoning + # rejects it; without reasoning it is accepted. + reasoning = resolve_reasoning( + "openrouter", "openai/gpt-5.5", "api_key", C.PROVIDER_DEFAULT + ) + assert reasoning.level is None and reasoning.omit_temperature + off = resolve_reasoning("openrouter", "openai/gpt-5.5", "api_key", C.OFF) + assert off.off and not off.omit_temperature def test_provider_default_sends_the_documented_level_where_one_is_required(): @@ -498,9 +539,10 @@ def test_codex_translator_uses_the_resolved_effort(): {"model": "gpt-5.5", "messages": messages, "reasoning_effort": "high"}, "k" ) assert ruled["reasoning"] == {"effort": "high", "summary": "auto"} - # No rule: the backend still requires the block, so today's medium stays. - unruled = _translate_request({"model": "gpt-5.4", "messages": messages}, "k") - assert unruled["reasoning"] == {"effort": "medium", "summary": "auto"} + # No reasoning_effort (VLM requests send none): the backend still + # requires the block, so it gets the Codex default. + unset = _translate_request({"model": "gpt-6.1-sol", "messages": messages}, "k") + assert unset["reasoning"] == {"effort": "medium", "summary": "auto"} # ────────────────────── 4. transports at CraftBot's settings ───────────────── From fb3f785edd2e9a5ecf5c88ec950d5174c477f2c1 Mon Sep 17 00:00:00 2001 From: CraftBot Date: Fri, 2 Oct 2026 10:38:57 +0900 Subject: [PATCH 5/5] fix duplicated steps and second decision per request --- agent_core/core/impl/llm/cache/gemini.py | 4 +- agent_core/core/impl/llm/interface.py | 132 ++++++++++-------- agent_core/core/impl/llm/reasoning_wire.py | 70 +++++++++- .../core/impl/llm/transports/__init__.py | 5 +- .../impl/llm/transports/anthropic_messages.py | 29 ++-- .../impl/llm/transports/bedrock_converse.py | 37 +++-- .../impl/llm/transports/byteplus_responses.py | 10 +- .../impl/llm/transports/chat_completions.py | 44 +++--- .../core/impl/llm/transports/gemini_native.py | 35 ++--- tests/llm/test_reasoning_rules.py | 66 ++++++++- tests/test_bedrock_token_normalization.py | 5 +- tests/test_context_budget.py | 4 +- 12 files changed, 291 insertions(+), 150 deletions(-) diff --git a/agent_core/core/impl/llm/cache/gemini.py b/agent_core/core/impl/llm/cache/gemini.py index 1b1376a9..f6779306 100644 --- a/agent_core/core/impl/llm/cache/gemini.py +++ b/agent_core/core/impl/llm/cache/gemini.py @@ -78,7 +78,7 @@ def get_or_create_cache( system_prompt: str, user_prompt: str, call_type: str, - temperature: float, + temperature: Optional[float], max_tokens: int, thinking_budget: Optional[int] = None, thinking_level: Optional[str] = None, @@ -89,7 +89,7 @@ def get_or_create_cache( system_prompt: The system prompt to cache. user_prompt: The user prompt for this request. call_type: Type of LLM call (e.g., "reasoning", "action_selection"). - temperature: Sampling temperature. + temperature: Sampling temperature (None: not sent). max_tokens: Maximum output tokens. thinking_budget: Reasoning token budget (Gemini 2.5), forwarded to every generation call. diff --git a/agent_core/core/impl/llm/interface.py b/agent_core/core/impl/llm/interface.py index ff75639b..9e3d9b2c 100644 --- a/agent_core/core/impl/llm/interface.py +++ b/agent_core/core/impl/llm/interface.py @@ -17,7 +17,7 @@ import contextvars import re import time -from typing import Any, Dict, List, Optional +from typing import Any, Dict, List, Optional, Set, Tuple from agent_core.decorators import profile, OperationCategory @@ -69,15 +69,6 @@ "_llm_call_ctx", default={} ) -# Per-call metadata (prompt identity + start time) propagated from the public -# entry methods down to the capture chokepoint (_call_log_to_db) without -# threading it through every provider method. asyncio.to_thread copies the -# context into the worker thread, so this survives the sync offload, and each -# asyncio Task / thread gets its own copy so concurrent calls don't clobber. -_llm_call_ctx: contextvars.ContextVar[dict] = contextvars.ContextVar( - "_llm_call_ctx", default={} -) - # Session key of the session call in flight. Set by the public session entry # points and read by _report_usage_async, so the input count the provider @@ -236,9 +227,12 @@ def __init__( # multi-provider outage terminates instead of nesting # primary -> fb -> fb-of-fb recursion. self._is_fallback_instance = False - # (provider, model, auth_mode) whose reasoning default was last - # logged, so the decision is logged once per model, not per call. - self._reasoning_logged_for: Optional[tuple] = None + # (provider, model, auth_mode, choice) combinations whose reasoning + # decision has been logged: each is logged once, not per call, even + # while concurrent sessions alternate between different choices. + self._reasoning_logged: Set[ + Tuple[Optional[str], Optional[str], str, ReasoningChoice] + ] = set() # Defer imports to avoid circular dependency from app.models.factory import ModelFactory @@ -783,17 +777,20 @@ def _check_context_fits( system_prompt: Optional[str], user_prompt: Optional[str] = None, messages: Optional[List[dict]] = None, + *, + reasoning: Optional[ReasoningDecision], ) -> None: """Refuse a request that cannot fit the configured context window. Counts the payload that is about to be sent: the system prompt plus either the single user prompt or every accumulated message. Input and - the output reservation share the window. + the output reservation share the window. ``reasoning`` is the + request's decision, the same one its transport sends. """ from app.config import get_context_window window = get_context_window() - reserved = self._output_reservation() + reserved = self._output_reservation(reasoning) budget = window - reserved total = count_tokens(system_prompt or "") @@ -853,18 +850,18 @@ def reasoning_decision(self) -> Optional[ReasoningDecision]: """What the call in flight sends for reasoning. None means the model has no rule in agent_core/core/models/reasoning.py - and its requests carry no reasoning parameter. Resolved on every call - so a model switch (reinitialize), a fallback interface, or a changed - session choice applies from the next request; logged once per - distinct model and choice. + and its requests carry no reasoning parameter. Each request resolves + it once, just before its context-window check, and hands the same + decision to that check and to its transport; so a model switch + (reinitialize), a fallback interface, or a changed session choice + applies from the next request. Logged once per distinct model and + choice. """ choice = self.reasoning_choice() - decision = resolve_reasoning( - self.provider, self.model, self._auth_mode, choice - ) + decision = resolve_reasoning(self.provider, self.model, self._auth_mode, choice) log_key = (self.provider, self.model, self._auth_mode, choice) - if log_key != self._reasoning_logged_for: - self._reasoning_logged_for = log_key + if log_key not in self._reasoning_logged: + self._reasoning_logged.add(log_key) if decision is None: logger.info( f"[REASONING] {self.provider}/{self.model}: no reasoning rule " @@ -874,17 +871,17 @@ def reasoning_decision(self) -> Optional[ReasoningDecision]: logger.info(f"[REASONING] {decision.key}: {decision.describe()}") return decision - def _output_reservation(self) -> int: - """Output tokens a request reserves out of the context window. + def _output_reservation(self, reasoning: Optional[ReasoningDecision]) -> int: + """Output tokens a request with this reasoning decision reserves out + of the context window. - Requests with a reasoning default carry a larger output cap (reasoning - tokens count against it), so the window check reserves that cap; - otherwise the provider would reject the request for size instead. + A reasoning request carries a larger output cap (reasoning tokens + count against it), so the window check reserves that cap; otherwise + the provider would reject the request for size instead. """ - decision = self.reasoning_decision() - if decision is None: + if reasoning is None: return self.max_tokens - return decision.output_cap(self.max_tokens) + return reasoning.output_cap(self.max_tokens) def _generate_response_sync( self, @@ -931,8 +928,15 @@ def _generate_response_sync( ) if _transport is None: # pragma: no cover raise RuntimeError(f"Unknown provider {self.provider!r}") - self._check_context_fits(system_prompt, user_prompt) - response = _transport(self, system_prompt, user_prompt, json_mode=json_mode) + reasoning = self.reasoning_decision() + self._check_context_fits(system_prompt, user_prompt, reasoning=reasoning) + response = _transport( + self, + system_prompt, + user_prompt, + json_mode=json_mode, + reasoning=reasoning, + ) content = response.get("content", "").strip() # Check if response is empty and provide diagnostics @@ -1237,7 +1241,7 @@ def fits_context( else: projected = last + count_tokens(pending) return ( - projected + self._output_reservation() + projected + self._output_reservation(self.reasoning_decision()) <= get_context_window() - get_reserve_tokens() ) @@ -1483,12 +1487,16 @@ def _generate_response_with_session_sync( f"sending {len(contents)} total contents" ) - self._check_context_fits(effective_system_prompt, messages=contents) + reasoning = self.reasoning_decision() + self._check_context_fits( + effective_system_prompt, messages=contents, reasoning=reasoning + ) response = self._generate_gemini( effective_system_prompt, user_prompt, call_type=call_type, contents_override=contents, + reasoning=reasoning, ) assistant_content = response.get("content", "") @@ -1545,12 +1553,16 @@ def _generate_response_with_session_sync( f"{len(history)} history msgs, sending {len(or_messages)} total" ) - self._check_context_fits(None, messages=or_messages) + reasoning = self.reasoning_decision() + self._check_context_fits( + None, messages=or_messages, reasoning=reasoning + ) response = self._generate_openai( effective_system_prompt, user_prompt, call_type=call_type, messages_override=or_messages, + reasoning=reasoning, ) assistant_content = response.get("content", "") @@ -1587,12 +1599,16 @@ def _generate_response_with_session_sync( f"{len(history)} history msgs, sending {len(oa_messages)} total" ) - self._check_context_fits(None, messages=oa_messages) + reasoning = self.reasoning_decision() + self._check_context_fits( + None, messages=oa_messages, reasoning=reasoning + ) response = self._generate_openai( effective_system_prompt, user_prompt, call_type=call_type, messages_override=oa_messages, + reasoning=reasoning, ) assistant_content = response.get("content", "") @@ -1667,12 +1683,16 @@ def _generate_response_with_session_sync( ) # Call Anthropic with the full multi-turn messages - self._check_context_fits(effective_system_prompt, messages=messages) + reasoning = self.reasoning_decision() + self._check_context_fits( + effective_system_prompt, messages=messages, reasoning=reasoning + ) response = self._generate_anthropic( effective_system_prompt, user_prompt, call_type=call_type, messages=messages, + reasoning=reasoning, ) # On success, accumulate the user message + assistant response in history @@ -1740,12 +1760,16 @@ def _generate_response_with_session_sync( f"sending {len(messages)} msgs to Converse" ) - self._check_context_fits(effective_system_prompt, messages=messages) + reasoning = self.reasoning_decision() + self._check_context_fits( + effective_system_prompt, messages=messages, reasoning=reasoning + ) response = self._generate_bedrock( effective_system_prompt, user_prompt, call_type=call_type, messages=messages, + reasoning=reasoning, ) # On success, accumulate the user message + assistant response in @@ -2044,6 +2068,8 @@ def _generate_openai( call_type: Optional[str] = None, messages_override: Optional[List[Dict[str, Any]]] = None, json_mode: bool = True, + *, + reasoning: Optional[ReasoningDecision], ) -> Dict[str, Any]: """Delegate to the chat_completions transport (Phase 2).""" return _transports.chat_completions.generate_openai( @@ -2053,15 +2079,7 @@ def _generate_openai( call_type=call_type, messages_override=messages_override, json_mode=json_mode, - ) - - @profile("llm_ollama_call", OperationCategory.LLM) - def _generate_ollama( - self, system_prompt: str | None, user_prompt: str, json_mode: bool = True - ) -> Dict[str, Any]: - """Delegate to the chat_completions transport's Ollama path (Phase 2).""" - return _transports.chat_completions.generate_ollama( - self, system_prompt, user_prompt, json_mode=json_mode + reasoning=reasoning, ) @profile("llm_gemini_call", OperationCategory.LLM) @@ -2072,6 +2090,8 @@ def _generate_gemini( call_type: Optional[str] = None, contents_override: Optional[List[Dict[str, Any]]] = None, json_mode: bool = True, + *, + reasoning: Optional[ReasoningDecision], ) -> Dict[str, Any]: """Delegate to the gemini_native transport (Phase 2).""" return _transports.gemini_native.generate( @@ -2081,15 +2101,9 @@ def _generate_gemini( call_type=call_type, contents_override=contents_override, json_mode=json_mode, + reasoning=reasoning, ) - @profile("llm_byteplus_call", OperationCategory.LLM) - def _generate_byteplus( - self, system_prompt: str | None, user_prompt: str - ) -> Dict[str, Any]: - """Delegate to the byteplus_responses transport (Phase 2).""" - return _transports.byteplus_responses.generate(self, system_prompt, user_prompt) - def _parse_responses_api_content(self, result: Dict[str, Any]) -> str: """Parse content from BytePlus Responses API response. @@ -2119,6 +2133,8 @@ def _generate_anthropic( user_prompt: str, call_type: Optional[str] = None, messages: Optional[List[dict]] = None, + *, + reasoning: Optional[ReasoningDecision], ) -> Dict[str, Any]: """Delegate to the anthropic_messages transport (Phase 2).""" return _transports.anthropic_messages.generate( @@ -2127,6 +2143,7 @@ def _generate_anthropic( user_prompt, call_type=call_type, messages=messages, + reasoning=reasoning, ) # ─────────── Bedrock model capability detection ─────────────────── @@ -2152,6 +2169,8 @@ def _generate_bedrock( user_prompt: str, call_type: Optional[str] = None, messages: Optional[List[dict]] = None, + *, + reasoning: Optional[ReasoningDecision], ) -> Dict[str, Any]: """Delegate to the bedrock_converse transport (Phase 2).""" return _transports.bedrock_converse.generate( @@ -2160,6 +2179,7 @@ def _generate_bedrock( user_prompt, call_type=call_type, messages=messages, + reasoning=reasoning, ) # ─────────────────── CLI helper for ad‑hoc testing ─────────────────── diff --git a/agent_core/core/impl/llm/reasoning_wire.py b/agent_core/core/impl/llm/reasoning_wire.py index 32d9189e..addbff2b 100644 --- a/agent_core/core/impl/llm/reasoning_wire.py +++ b/agent_core/core/impl/llm/reasoning_wire.py @@ -1,12 +1,18 @@ # -*- coding: utf-8 -*- -"""Render a reasoning decision into request fields, one function per transport. +"""Apply a reasoning decision to one request, one function per transport. The table, the choices, and the level policy live in agent_core/core/models/reasoning.py; this module only knows how each -transport spells a ``ReasoningDecision``. Callers skip these functions -entirely when a model has no rule, so a model without a rule never gains a -field here. A decision that sends nothing (the provider default, or off on a -model whose default is no reasoning) renders as no fields. +transport carries a ``ReasoningDecision``. The interface resolves the +decision once per request and hands the same decision to its context-window +check and to the transport, whose ``for_*`` function below turns it into the +request's output cap, temperature permission, and reasoning fields. + +A request without a decision (the model has no rule) keeps the transport's +own cap and temperature and gains no field, so its payload is exactly the +one sent before per-model rules existed. A decision that sends nothing (the +provider default, or off on a model whose default is no reasoning) renders +as no fields. A decision whose wire the transport cannot express is a bug in the rules table (a row filed under the wrong provider, or an explicit off on a wire @@ -17,13 +23,65 @@ from __future__ import annotations -from typing import Any, Dict, Tuple +from dataclasses import dataclass +from typing import Any, Callable, Dict, Generic, Optional, Tuple, TypeVar from agent_core.core.models.reasoning import ReasoningDecision, ReasoningWire #: The disable value of the OpenAI-style effort parameter. EFFORT_OFF = "none" +FieldsT = TypeVar("FieldsT") + + +@dataclass(frozen=True) +class RequestReasoning(Generic[FieldsT]): + """How one request carries its reasoning decision on one transport.""" + + #: Output-token cap: the transport's own cap, raised while reasoning + #: (reasoning tokens count against it), never lowered. + output_cap: int + #: Whether the request may carry an explicit temperature. + send_temperature: bool + #: The transport's reasoning fields; empty when nothing is sent. + fields: FieldsT + + +def _for_request( + decision: Optional[ReasoningDecision], + base_cap: int, + render: Callable[[ReasoningDecision], FieldsT], + no_fields: FieldsT, +) -> RequestReasoning[FieldsT]: + if decision is None: + return RequestReasoning(base_cap, True, no_fields) + return RequestReasoning( + output_cap=decision.output_cap(base_cap), + send_temperature=not decision.omit_temperature, + fields=render(decision), + ) + + +def for_chat_completions( + decision: Optional[ReasoningDecision], base_cap: int +) -> RequestReasoning[Tuple[Dict[str, Any], Dict[str, Any]]]: + """Chat Completions; fields are (top-level kwargs, extra_body).""" + return _for_request(decision, base_cap, chat_completions_fields, ({}, {})) + + +def for_anthropic( + decision: Optional[ReasoningDecision], base_cap: int +) -> RequestReasoning[Dict[str, Any]]: + """Anthropic Messages, and Claude on Bedrock Converse.""" + return _for_request(decision, base_cap, anthropic_fields, {}) + + +def for_gemini( + decision: Optional[ReasoningDecision], base_cap: int +) -> RequestReasoning[Dict[str, Any]]: + """Gemini generateContent; fields are GeminiClient keyword arguments.""" + return _for_request(decision, base_cap, gemini_thinking_kwargs, {}) + def _unsupported(decision: ReasoningDecision, transport: str) -> ValueError: what = "an explicit off" if decision.off else f"wire {decision.wire.value!r}" diff --git a/agent_core/core/impl/llm/transports/__init__.py b/agent_core/core/impl/llm/transports/__init__.py index 6646e72f..16aff246 100644 --- a/agent_core/core/impl/llm/transports/__init__.py +++ b/agent_core/core/impl/llm/transports/__init__.py @@ -8,8 +8,11 @@ session state stays on LLMInterface (NFR-3 in docs/PROVIDER_LAYER_CATCHUP.md). TRANSPORTS maps ProviderProfile.wire -> the transport's generate callable -with signature (iface, system_prompt, user_prompt, json_mode=True) -> response dict +with signature (iface, system_prompt, user_prompt, json_mode=True, *, +reasoning) -> response dict ({"content", "tokens_used", "cached_tokens"?, "error"?, "error_info_obj"?}). +``reasoning`` is the request's ReasoningDecision, resolved once by the +interface (None when the model has no reasoning rule). Session-mode entry points with richer signatures are exposed as module functions and called by the session dispatcher on LLMInterface. """ diff --git a/agent_core/core/impl/llm/transports/anthropic_messages.py b/agent_core/core/impl/llm/transports/anthropic_messages.py index 56639b0f..abafa6fd 100644 --- a/agent_core/core/impl/llm/transports/anthropic_messages.py +++ b/agent_core/core/impl/llm/transports/anthropic_messages.py @@ -13,6 +13,7 @@ from agent_core.core.impl.llm import reasoning_wire from agent_core.core.impl.llm.cache import get_cache_config, get_cache_metrics from agent_core.core.impl.llm.errors import classify_llm_error +from agent_core.core.models.reasoning import ReasoningDecision from agent_core.utils.logger import logger # Anthropic requires max_tokens; 16384 (Claude 4 default) avoids truncation. @@ -27,6 +28,8 @@ def generate( call_type: Optional[str] = None, messages: Optional[List[dict]] = None, json_mode: bool = True, + *, + reasoning: Optional[ReasoningDecision], ) -> Dict[str, Any]: """Generate response using Anthropic with prompt caching. @@ -53,6 +56,9 @@ def generate( When provided, uses extended 1-hour TTL for better cache hit rates. messages: Optional pre-built messages list for multi-turn sessions. When provided, used instead of building a single-turn message. + reasoning: This request's reasoning decision, resolved once by the + interface (None: the model has no rule, so the request is + shaped exactly as it was before per-model rules). Cache hits are logged when `cache_read_input_tokens` > 0 in the response. """ @@ -75,20 +81,14 @@ def generate( if not iface._anthropic_client: raise RuntimeError("Anthropic client was not initialised.") - # Reasoning default for this exact model (None: no rule, so the - # request is shaped exactly as it was before reasoning defaults). - reasoning = iface.reasoning_decision() + # Claude rows' reasoning caps stay below the SDK's non-streaming + # ceiling at every level. + applied = reasoning_wire.for_anthropic(reasoning, _DEFAULT_MAX_TOKENS) # Build the message - use pre-built messages for multi-turn, or single-turn. - # Thinking tokens count against max_tokens, so a reasoning default - # raises it (staying below the SDK's non-streaming ceiling). message_kwargs: Dict[str, Any] = { "model": iface.model, - "max_tokens": ( - _DEFAULT_MAX_TOKENS - if reasoning is None - else reasoning.output_cap(_DEFAULT_MAX_TOKENS) - ), + "max_tokens": applied.output_cap, "messages": messages if messages is not None else [ @@ -122,13 +122,12 @@ def generate( # Short prompt - use simple string format (no caching) message_kwargs["system"] = system_prompt - if reasoning is not None: - message_kwargs.update(reasoning_wire.anthropic_fields(reasoning)) + message_kwargs.update(applied.fields) # Thinking is incompatible with temperature on Claude 4.5/4.6, and - # Claude 4.7+ rejects any non-default temperature, so rows with - # reasoning drop it (the model then uses its default). - if reasoning is None or not reasoning.omit_temperature: + # Claude 4.7+ rejects any non-default temperature, so those requests + # drop it (the model then uses its default). + if applied.send_temperature: message_kwargs["extra_body"] = {"temperature": iface.temperature} response = iface._anthropic_client.messages.create(**message_kwargs) diff --git a/agent_core/core/impl/llm/transports/bedrock_converse.py b/agent_core/core/impl/llm/transports/bedrock_converse.py index 7c41d07d..3af9884b 100644 --- a/agent_core/core/impl/llm/transports/bedrock_converse.py +++ b/agent_core/core/impl/llm/transports/bedrock_converse.py @@ -14,6 +14,7 @@ from agent_core.core.impl.llm import reasoning_wire from agent_core.core.impl.llm.cache import get_cache_config, get_cache_metrics from agent_core.core.impl.llm.errors import classify_llm_error +from agent_core.core.models.reasoning import ReasoningDecision from agent_core.utils.logger import logger @@ -25,6 +26,8 @@ def generate( call_type: Optional[str] = None, messages: Optional[List[dict]] = None, json_mode: bool = True, + *, + reasoning: Optional[ReasoningDecision], ) -> Dict[str, Any]: """Generate response via AWS Bedrock Converse API with prompt caching. @@ -47,6 +50,9 @@ def generate( needed and placing it in messages lets the cache grow with the conversation). When messages is None, falls back to a fresh single-turn call with cachePoint on the system block. + reasoning: This request's reasoning decision, resolved once by the + interface (None: the model has no rule, so the request is shaped + exactly as it was before per-model rules). """ token_count_input = token_count_output = 0 total_tokens = 0 @@ -71,31 +77,22 @@ def generate( else [{"role": "user", "content": [{"text": user_prompt}]}] ) - # Reasoning default for this exact model (None: no rule, so the - # request is shaped exactly as it was before reasoning defaults). - reasoning = iface.reasoning_decision() + # Claude on Converse takes the Messages-API thinking/effort keys + # through additionalModelRequestFields, with the same cap and + # temperature consequences. + applied = reasoning_wire.for_anthropic(reasoning, iface.max_tokens) + inference_config: Dict[str, Any] = {} + if applied.send_temperature: + inference_config["temperature"] = iface.temperature + inference_config["maxTokens"] = applied.output_cap converse_kwargs: Dict[str, Any] = { "modelId": iface.model, "messages": converse_messages, - "inferenceConfig": { - "temperature": iface.temperature, - "maxTokens": iface.max_tokens, - }, + "inferenceConfig": inference_config, } - if reasoning is not None: - # Claude on Converse takes the Messages-API thinking/effort keys - # through additionalModelRequestFields. Thinking tokens count - # against maxTokens, and thinking is incompatible with (4.5/4.6) - # or rejects (4.7+) a non-default temperature. - reasoning_fields = reasoning_wire.anthropic_fields(reasoning) - if reasoning_fields: - converse_kwargs["additionalModelRequestFields"] = reasoning_fields - converse_kwargs["inferenceConfig"]["maxTokens"] = reasoning.output_cap( - iface.max_tokens - ) - if reasoning.omit_temperature: - del converse_kwargs["inferenceConfig"]["temperature"] + if applied.fields: + converse_kwargs["additionalModelRequestFields"] = applied.fields if system_prompt: # When messages already carry a cachePoint (multi-turn first diff --git a/agent_core/core/impl/llm/transports/byteplus_responses.py b/agent_core/core/impl/llm/transports/byteplus_responses.py index 5e726aae..ea5db649 100644 --- a/agent_core/core/impl/llm/transports/byteplus_responses.py +++ b/agent_core/core/impl/llm/transports/byteplus_responses.py @@ -22,12 +22,18 @@ get_cache_metrics, ) from agent_core.core.impl.llm.errors import classify_llm_error +from agent_core.core.models.reasoning import ReasoningDecision from agent_core.utils.logger import logger @profile("llm_byteplus_call", OperationCategory.LLM) def generate( - iface, system_prompt: str | None, user_prompt: str, json_mode: bool = True + iface, + system_prompt: str | None, + user_prompt: str, + json_mode: bool = True, + *, + reasoning: Optional[ReasoningDecision], ) -> Dict[str, Any]: """Generate response using BytePlus with automatic prefix caching. @@ -35,6 +41,8 @@ def generate( ``json_mode`` is accepted for transport-signature uniformity but unused: the Responses API has no json_object knob here — JSON is prompt-instructed. + ``reasoning`` is accepted for the same reason: no BytePlus model has a + reasoning rule, so it is always None. """ config = get_cache_config() # Use prefix caching if: diff --git a/agent_core/core/impl/llm/transports/chat_completions.py b/agent_core/core/impl/llm/transports/chat_completions.py index f0871153..c58977e1 100644 --- a/agent_core/core/impl/llm/transports/chat_completions.py +++ b/agent_core/core/impl/llm/transports/chat_completions.py @@ -22,6 +22,7 @@ from agent_core.core.impl.llm import reasoning_wire from agent_core.core.impl.llm.cache import get_cache_config, get_cache_metrics from agent_core.core.impl.llm.errors import classify_llm_error, provider_display_name +from agent_core.core.models.reasoning import ReasoningDecision from agent_core.core.models.registry import ( get_registry as _get_registry, supports_prompt_cache_key as _supports_pck, @@ -51,6 +52,8 @@ def generate_openai( call_type: Optional[str] = None, messages_override: Optional[List[Dict[str, Any]]] = None, json_mode: bool = True, + *, + reasoning: Optional[ReasoningDecision], ) -> Dict[str, Any]: """Generate response using OpenAI with automatic prompt caching. @@ -73,6 +76,9 @@ def generate_openai( the accumulating prefix via OR's cache_control field. When set, it's sent verbatim — system_prompt is still passed in for cache- key derivation but the request body uses messages_override. + reasoning: This request's reasoning decision, resolved once by the + interface (None: the model has no rule, so the request is shaped + exactly as it was before per-model rules). Cache hits are logged when cached_tokens > 0 in the response. """ @@ -115,14 +121,10 @@ def generate_openai( "model": iface.model, "messages": messages, } - # Reasoning default for this exact model (None: no rule, so the - # request is shaped exactly as it was before reasoning defaults). - reasoning = iface.reasoning_decision() + applied = reasoning_wire.for_chat_completions(reasoning, iface.max_tokens) _profile = _get_registry().get(iface.provider) _temp = _resolve_temperature(_profile, iface.temperature) - if _temp is not _OMIT_TEMPERATURE and not ( - reasoning is not None and reasoning.omit_temperature - ): + if _temp is not _OMIT_TEMPERATURE and applied.send_temperature: request_kwargs["temperature"] = _temp # Output tokens: cap the VALUE to the provider's output limit (several @@ -130,13 +132,8 @@ def generate_openai( # when it's exceeded), and pick the FIELD NAME per provider policy # (profile.uses_max_completion_tokens: OpenAI/Cerebras/MiniMax/Groq # take 'max_completion_tokens'; everyone else legacy 'max_tokens'). - # Reasoning tokens count against this cap, so a reasoning default - # raises it (never lowers it) to leave room for the answer. - _max_tokens_value = ( - iface.max_tokens - if reasoning is None - else reasoning.output_cap(iface.max_tokens) - ) + # The value starts from the reasoning-raised cap. + _max_tokens_value = applied.output_cap if _profile is not None and _profile.max_output_tokens: _max_tokens_value = min(_max_tokens_value, _profile.max_output_tokens) uses_max_completion_tokens = ( @@ -211,12 +208,9 @@ def generate_openai( f"[OPENROUTER] Anthropic cache_control: {cache_control} (model={iface.model})" ) - if reasoning is not None: - reasoning_top, reasoning_extra = reasoning_wire.chat_completions_fields( - reasoning - ) - request_kwargs.update(reasoning_top) - extra_body.update(reasoning_extra) + reasoning_top, reasoning_extra = applied.fields + request_kwargs.update(reasoning_top) + extra_body.update(reasoning_extra) if extra_body: request_kwargs["extra_body"] = extra_body @@ -338,8 +332,18 @@ def generate_openai( @profile("llm_ollama_call", OperationCategory.LLM) def generate_ollama( - iface, system_prompt: str | None, user_prompt: str, json_mode: bool = True + iface, + system_prompt: str | None, + user_prompt: str, + json_mode: bool = True, + *, + reasoning: Optional[ReasoningDecision], ) -> Dict[str, Any]: + """Generate a response through Ollama's native ``/api/generate``. + + ``reasoning`` is accepted for transport-signature uniformity but unused: + no Ollama-served model has a reasoning rule, so it is always None. + """ token_count_input = token_count_output = 0 total_tokens = 0 status = "failed" diff --git a/agent_core/core/impl/llm/transports/gemini_native.py b/agent_core/core/impl/llm/transports/gemini_native.py index 4ddc8305..251c965d 100644 --- a/agent_core/core/impl/llm/transports/gemini_native.py +++ b/agent_core/core/impl/llm/transports/gemini_native.py @@ -13,6 +13,7 @@ from agent_core.core.impl.llm import reasoning_wire from agent_core.core.impl.llm.cache import get_cache_config, get_cache_metrics from agent_core.core.impl.llm.errors import classify_llm_error +from agent_core.core.models.reasoning import ReasoningDecision from agent_core.utils.logger import logger @@ -24,6 +25,8 @@ def generate( call_type: Optional[str] = None, contents_override: Optional[List[Dict[str, Any]]] = None, json_mode: bool = True, + *, + reasoning: Optional[ReasoningDecision], ) -> Dict[str, Any]: """Generate response using Gemini with explicit or implicit caching. @@ -45,6 +48,9 @@ def generate( history so Gemini's implicit caching catches the growing stable prefix automatically (caching covers more tokens with every turn without us needing to manage a named cache object). + reasoning: This request's reasoning decision, resolved once by the + interface (None: the model has no rule, so Gemini's own default + thinking is left untouched). Returns: Dict with tokens_used, content, cached_tokens. @@ -53,24 +59,19 @@ def generate( # Per-call reasoning cap, set by callers that pass thinking_budget (e.g. the # entity-judge pipeline). Rides the shared per-call context so no transport - # signature changes. An explicit per-call budget wins over the model's - # reasoning default; otherwise the default for this exact model applies - # (None: no rule, so Gemini's own default thinking is left untouched). + # signature changes. An explicit per-call budget replaces the request's + # reasoning decision. from agent_core.core.impl.llm.interface import _llm_call_ctx caller_budget = (_llm_call_ctx.get() or {}).get("thinking_budget") - reasoning = iface.reasoning_decision() if caller_budget is None else None - if caller_budget is not None: - thinking: Dict[str, Any] = {"thinking_budget": caller_budget} - elif reasoning is not None: - thinking = reasoning_wire.gemini_thinking_kwargs(reasoning) - else: - thinking = {} - # Thinking tokens count against maxOutputTokens, so a reasoning default - # raises the cap (never lowers it) to leave room for the answer. - max_output_tokens = ( - iface.max_tokens if reasoning is None else reasoning.output_cap(iface.max_tokens) + applied = reasoning_wire.for_gemini( + reasoning if caller_budget is None else None, iface.max_tokens + ) + thinking: Dict[str, Any] = ( + applied.fields if caller_budget is None else {"thinking_budget": caller_budget} ) + temperature = iface.temperature if applied.send_temperature else None + max_output_tokens = applied.output_cap token_count_input = token_count_output = 0 cached_tokens = 0 @@ -98,7 +99,7 @@ def generate( iface.model, contents=contents_override, system_prompt=system_prompt, - temperature=iface.temperature, + temperature=temperature, max_output_tokens=max_output_tokens, json_mode=json_mode, **thinking, @@ -130,7 +131,7 @@ def generate( system_prompt=system_prompt, user_prompt=user_prompt, call_type=call_type, - temperature=iface.temperature, + temperature=temperature, max_tokens=max_output_tokens, **thinking, ) @@ -140,7 +141,7 @@ def generate( iface.model, prompt=user_prompt, system_prompt=system_prompt, - temperature=iface.temperature, + temperature=temperature, max_output_tokens=max_output_tokens, json_mode=json_mode, **thinking, diff --git a/tests/llm/test_reasoning_rules.py b/tests/llm/test_reasoning_rules.py index 78abecc8..ff942398 100644 --- a/tests/llm/test_reasoning_rules.py +++ b/tests/llm/test_reasoning_rules.py @@ -746,7 +746,7 @@ def test_gemini_caller_budget_wins_over_the_session(monkeypatch, bound_session): assert config["maxOutputTokens"] == APP_MAX_TOKENS -def test_context_check_reserves_the_reasoning_cap(monkeypatch, bound_session): +def test_context_check_reserves_the_requests_reasoning_cap(monkeypatch, bound_session): import app.config as app_config monkeypatch.setattr(app_config, "get_context_window", lambda: 40_000) @@ -754,19 +754,71 @@ def test_context_check_reserves_the_reasoning_cap(monkeypatch, bound_session): unruled, _ = build_interface(monkeypatch, "openai", "gpt-4o") unruled.max_tokens = APP_MAX_TOKENS - assert unruled._output_reservation() == APP_MAX_TOKENS - unruled._check_context_fits(None, prompt) + assert unruled._output_reservation(unruled.reasoning_decision()) == APP_MAX_TOKENS + unruled._check_context_fits(None, prompt, reasoning=unruled.reasoning_decision()) ruled, _ = build_interface(monkeypatch, "openai", "gpt-5.2-2025-12-11") ruled.max_tokens = APP_MAX_TOKENS - assert ruled._output_reservation() == 32_000 + high = ruled.reasoning_decision() + assert ruled._output_reservation(high) == 32_000 with pytest.raises(LLMContextOverflowError, match="32000 reserved for output"): - ruled._check_context_fits(None, prompt) + ruled._check_context_fits(None, prompt, reasoning=high) + # The request path refuses it before anything is sent. + with pytest.raises(LLMContextOverflowError): + ruled.generate_response(system_prompt=None, user_prompt=prompt) # A session that turned reasoning off reserves only the caller's cap. with bound_session(C.OFF): - assert ruled._output_reservation() == APP_MAX_TOKENS - ruled._check_context_fits(None, prompt) + off = ruled.reasoning_decision() + assert ruled._output_reservation(off) == APP_MAX_TOKENS + ruled._check_context_fits(None, prompt, reasoning=off) + + +def test_each_request_resolves_its_reasoning_once(monkeypatch): + # The context check and the transport share one decision, so a choice + # changed mid-request cannot make them disagree. + iface, rec = build_interface(monkeypatch, "openai", "gpt-5.2-2025-12-11") + iface.max_tokens = APP_MAX_TOKENS + resolutions: List[ReasoningDecision] = [] + resolve = iface.reasoning_decision + + def counting_resolve(): + resolutions.append(resolve()) + return resolutions[-1] + + monkeypatch.setattr(iface, "reasoning_decision", counting_resolve) + iface.generate_response(system_prompt=GOLDEN_SYSTEM_PROMPT, user_prompt="hi") + assert len(resolutions) == 1 + + iface.create_session_cache("task", "action_selection", GOLDEN_SYSTEM_PROMPT) + iface.generate_response_with_session( + "task", "action_selection", "hi", log_response=False + ) + assert len(resolutions) == 2 + assert [call["payload"]["reasoning_effort"] for call in rec.calls] == [ + "high", + "high", + ] + + +def test_each_model_and_choice_is_logged_once(monkeypatch, bound_session): + import agent_core.core.impl.llm.interface as interface_module + + iface, _ = build_interface(monkeypatch, "openai", "gpt-5.2-2025-12-11") + with bound_session(C.HIGH, session_id="s-high"): + pass + with bound_session(C.LOW, session_id="s-low"): + pass + logged: List[str] = [] + monkeypatch.setattr(interface_module, "logger", SimpleNamespace(info=logged.append)) + # Two sessions alternating, as concurrent sessions do. + for _ in range(3): + with StateSession.bind("s-high"): + iface.reasoning_decision() + with StateSession.bind("s-low"): + iface.reasoning_decision() + assert len(logged) == 2 + assert "choice=high" in logged[0] and "choice=low" in logged[1] def test_fold_check_reserves_the_same_cap_as_the_pre_send_check( diff --git a/tests/test_bedrock_token_normalization.py b/tests/test_bedrock_token_normalization.py index 58ce2a9d..2aa18a17 100644 --- a/tests/test_bedrock_token_normalization.py +++ b/tests/test_bedrock_token_normalization.py @@ -52,11 +52,8 @@ def test_llm_bedrock_input_includes_cache_and_cached_is_reads_only(): iface = LLMInterface.__new__(LLMInterface) reported = _stub_common(iface) iface._call_log_to_db = lambda *a, **kw: None - # Read by the per-model reasoning lookup every transport performs. - iface._auth_mode = "api_key" - iface._reasoning_logged_for = None - result = iface._generate_bedrock(None, "hi") + result = iface._generate_bedrock(None, "hi", reasoning=None) assert "error" not in result # input = 100 + 900 + 30, the full prompt; cached = reads only diff --git a/tests/test_context_budget.py b/tests/test_context_budget.py index a8ebf7a6..5c3ad86f 100644 --- a/tests/test_context_budget.py +++ b/tests/test_context_budget.py @@ -334,7 +334,9 @@ def test_preflight_refuses_a_request_that_does_not_fit(): iface = _make() with patch.object(app_config, "get_settings", return_value=_settings()): with pytest.raises(LLMContextOverflowError): - iface._check_context_fits("system", "word " * 200_000) + iface._check_context_fits( + "system", "word " * 200_000, reasoning=iface.reasoning_decision() + ) assert iface._consecutive_failures == 0