From e7b1c052cde5243529958b8db142371a9f5abc30 Mon Sep 17 00:00:00 2001 From: Amplifier <240397093+microsoft-amplifier@users.noreply.github.com> Date: Tue, 15 Sep 2026 06:53:46 -0700 Subject: [PATCH 1/2] feat: max_tokens becomes a real cap; max_tokens_fallback takes over the fallback `max_tokens` was documented as "Maximum context size" but implemented as priority-4 fallback in `_calculate_budget` -- consulted only when a provider published no window. Orchestrators always pass a provider, so the knob was silently dead in production: the shipped foundation bundle's `max_tokens: 300000` had no effect on when compaction fired, in either direction. The cadence probe had to patch this module's source in-container to add `budget = min(budget, self.max_tokens)` before either of its arms would compact at all. This ships that patch as the supported contract, splitting one overloaded key into the two jobs it was doing badly: - `max_tokens` (default None) is now a CAP, applied as min(derived_budget, max_tokens) on every derivation path with no exceptions. None means no cap -- use the model's full window. A value above the model's own window is a no-op by construction, not an error. - `max_tokens_fallback` (default 200,000) takes over the fallback role. 200,000 is the smallest context window across the current generation of the three major vendors (Anthropic 200K base, OpenAI 272K default, Google ~1M); a guess that is too large overflows a request, one too small only compacts early. Default None makes this a no-op for every existing config until someone opts in. An invalid cap (0, negative, non-int) degrades to "no cap" with a warning rather than being honored into a budget <= 0, which `_should_compact`'s `budget > 0` guard would turn into silently disabled compaction -- matching how `compaction_notice_token_reserve` already handles a self-defeating value. Also fixes the usage log in `add_message`, which divided by `max_tokens` -- a number that was never the governing denominator and is now usually unset. It now reports against the budget the last request actually ran on. Tests: `test_compaction_trigger_provenance.py` inverted to pin the new contract in both directions -- 45,000 caps low enough to compact where 70,000 does not, and an unset cap leaves the provider budget untouched. Plus validation coverage for the refused-cap path. 166 passed, 1 xfailed. Generated with Amplifier Co-Authored-By: Amplifier <240397093+microsoft-amplifier@users.noreply.github.com> --- README.md | 96 +++++++--- amplifier_module_context_simple/__init__.py | 122 +++++++++++-- tests/test_compaction_trigger_provenance.py | 190 +++++++++++++++----- 3 files changed, 333 insertions(+), 75 deletions(-) diff --git a/README.md b/README.md index 5f55ee3..819a7b0 100644 --- a/README.md +++ b/README.md @@ -43,9 +43,14 @@ module = "context-simple" name = "simple" [contexts.config] +max_tokens = 500000 # Optional CAP; default None = use the model's full window +max_tokens_fallback = 200000 # Only used when the provider publishes no window max_tool_result_bytes = 131072 # Optional override; default is 128 KiB ``` +`max_tokens` is the knob to reach for when you want to use **less** context +than the model allows -- see [`max_tokens` is a cap](#max_tokens-is-a-cap). + ### Tool-result text ingress cap Before admission, a tool message's string content (including JSON serialized @@ -94,7 +99,7 @@ The SimpleContextManager uses **ephemeral compaction** - admitted internal message history. Ingress-clipped tool text is irreversible and is not retained in the canonical transcript; compaction remains view-only. -Compaction triggers when token usage reaches the configured threshold (default: 92% of the **effective budget** -- which is derived from the provider, *not* from `max_tokens`; see [Where the compaction trigger comes from](#where-the-compaction-trigger-comes-from)): +Compaction triggers when token usage reaches the configured threshold (default: 92% of the **effective budget** -- the provider's own window, capped by `max_tokens` if you set one; see [Where the compaction trigger comes from](#where-the-compaction-trigger-comes-from)): ### Protected Messages (Never Removed) @@ -156,36 +161,83 @@ The trigger is one multiplication: trigger = compact_threshold * effective_budget ``` -`effective_budget` comes from `_calculate_budget()`, in this priority order: +`effective_budget` is computed in two steps. + +**Step 1 -- derive what the model offers** (`_derive_budget()`), in priority order: -1. an explicit `token_budget=` argument (deprecated, rarely used); +1. an explicit `token_budget=` argument (the provider's own overflow-shrink retry); 2. `provider.get_model_info()` -> `context_window - 0.5 * max_output_tokens - 4096`; 3. `provider.get_info().defaults` -> the same formula; -4. **only if none of the above yields a window**: the configured `max_tokens`. +4. **only if none of the above yields a window**: `max_tokens_fallback`. + +**Step 2 -- cap it** at `max_tokens`, if one is configured: + +``` +effective_budget = min(derived_budget, max_tokens) +``` + +### `max_tokens` is a cap + +`max_tokens` defaults to **`None`**, meaning *no cap* -- use the whole window +the model allows. Set it only when you want to use **less** than the model +offers: + +```yaml +context: + module: context-simple + config: + max_tokens: 500000 # compact as if the window were 500k +``` + +The lower of the two always wins, so setting `max_tokens` **above** the +model's own window is a no-op by construction, not an error. The cap applies +on every derivation path with no exceptions -- `min()` can only lower a +budget, never raise it, so a cap can make compaction fire earlier but can +never overflow a request. + +### `max_tokens_fallback` is the other half + +`max_tokens_fallback` (default **200,000**) answers only when a provider +publishes no usable window at all -- it is a floor under a missing number, not +a cap. 200,000 is the smallest context window across the current generation of +the three major vendors (Anthropic 200K base, OpenAI 272K default, Google +~1M), chosen small because a guess that is too large overflows a request while +one that is too small merely compacts early. + +A provider landing on this branch is a bug in **that provider** -- the durable +fix is for it to publish its window, not to tune this number. Both keys are +overridable per session: + +```yaml +# ~/.amplifier/settings.yaml +overrides: + context-simple: + config: + max_tokens_fallback: 400000 +``` -### `max_tokens` is a fallback, not a cap +### History: this knob used to be dead -Orchestrators call `get_messages_for_request(provider=provider)`, so in -practice branch 2 or 3 always answers and **branch 4 is never reached**. -The `max_tokens` value in your bundle config (the shipped foundation bundle -sets `max_tokens: 300000`) therefore has **no effect on when compaction -fires**. Lowering it to compact sooner, or raising it to compact later, is a -no-op on the wire. +Before the cap existed, `max_tokens` was consulted **only** at step 1.4 -- +as a fallback. Orchestrators always call +`get_messages_for_request(provider=provider)`, so branch 2 or 3 always +answered and branch 4 was never reached: the `max_tokens` in your bundle +config had **no effect on when compaction fired**, in either direction. -This is a real trap, not a theoretical one. The cadence probe that produced -the numbers below could not move the trigger with config at all: its harness +That was a real trap, not a theoretical one. The cadence probe that produced +the numbers below could not move the trigger with config at all -- its harness had to patch this module's source in-container to add -`budget = min(budget, self.max_tokens)` before either of its arms would -compact in a bounded run. +`budget = min(budget, self.max_tokens)`, which is precisely the behavior this +module now ships as a supported option. -`tests/test_compaction_trigger_provenance.py` pins this behavior in both -directions -- same history and same config compacts with no provider and does -*not* compact with one -- so the trap fails a test rather than a measurement -run. +`tests/test_compaction_trigger_provenance.py` pins the current contract in +both directions: a cap below the provider window moves the trigger, an unset +cap leaves the provider budget untouched, and a cap above the window is a +no-op. -**To move the trigger, move `compact_threshold`.** It is the only shipped knob -that expresses "compact later" independently of the provider, and the old -value stays reachable: +**To move the trigger as a *fraction*, move `compact_threshold`.** It +expresses "compact later" independently of the provider, where `max_tokens` +expresses it as an absolute ceiling: ```yaml context: diff --git a/amplifier_module_context_simple/__init__.py b/amplifier_module_context_simple/__init__.py index 5f5136d..d96a193 100644 --- a/amplifier_module_context_simple/__init__.py +++ b/amplifier_module_context_simple/__init__.py @@ -51,6 +51,24 @@ logger = logging.getLogger(__name__) DEFAULT_MAX_TOOL_RESULT_BYTES = 128 * 1024 + +DEFAULT_MAX_TOKENS_FALLBACK = 200_000 +"""Budget used ONLY when the provider reports no usable context window. + +Calibrated to the SMALLEST context window across the current generation of +the three major vendors, because a fallback is a guess and the safe direction +for a guess is small -- an over-large budget overflows the request, an +under-large one only compacts earlier: + + Anthropic claude-opus-5 / sonnet-5 / haiku-4-5 200,000 (base; 1M opt-in) + OpenAI gpt-5.6, gpt-6-astra 272,000 (default; ~900K opt-in) + Google gemini-3.x 1,048,576 + +Reached only by providers that publish neither `get_model_info()` nor +`get_info().defaults["context_window"]` + `["max_output_tokens"]`. Overridable +per-session via config `max_tokens_fallback`; the right long-term fix for any +provider landing here is for THAT provider to publish its window. +""" _TOOL_RESULT_INGRESS_MARKER = ( "[tool-result truncated at ingress: original_text_utf8_bytes={original_text_utf8_bytes}; " "retrieve missing content using narrower read/query parameters; do not repeat " @@ -118,6 +136,28 @@ def _is_human_message(msg: dict[str, Any]) -> bool: ) +def _validated_max_tokens(raw: Any) -> int | None: + """Return a usable `max_tokens` cap, or None for 'no cap'. + + A nonsense cap is refused LOUDLY and treated as absent rather than + honored, matching how `compaction_notice_token_reserve` handles a + self-defeating value: a cap of 0 or a negative number would drive the + budget to <= 0, and `_should_compact`'s `budget > 0` guard would then + silently disable compaction entirely -- the opposite of what someone + setting a cap is asking for. + """ + if raw is None: + return None + if isinstance(raw, bool) or not isinstance(raw, int) or raw <= 0: + logger.warning( + f"context-simple: ignoring invalid max_tokens {raw!r} " + f"(expected a positive integer, or None for no cap); " + f"running uncapped at the model's own window" + ) + return None + return raw + + async def mount(coordinator: ModuleCoordinator, config: dict[str, Any] | None = None): """ Mount the simple context manager. @@ -125,7 +165,14 @@ async def mount(coordinator: ModuleCoordinator, config: dict[str, Any] | None = Args: coordinator: Module coordinator config: Optional configuration - - max_tokens: Maximum context size (default: 200,000) + - max_tokens: CAP on the effective token budget (default: None = + no cap, use everything the model allows). Set this only to use + LESS than the model's window. It is applied as + `min(model_derived_budget, max_tokens)`, so setting it ABOVE the + model's own window has no effect -- the lower of the two wins. + - max_tokens_fallback: Budget used only when the provider reports + no usable window (default: 200,000). Not a cap; see + DEFAULT_MAX_TOKENS_FALLBACK for how the number was chosen. - compact_threshold: Trigger compaction at this usage (default: 0.92) - target_usage: Compact down to this usage (default: 0.50) - protected_recent: Always protect last N% of messages (default: 0.30) @@ -164,8 +211,13 @@ async def mount(coordinator: ModuleCoordinator, config: dict[str, Any] | None = ) token_meter = TOKEN_METER_ESTIMATE + max_tokens = _validated_max_tokens(config.get("max_tokens")) + context = SimpleContextManager( - max_tokens=config.get("max_tokens", 200_000), + max_tokens=max_tokens, + max_tokens_fallback=config.get( + "max_tokens_fallback", DEFAULT_MAX_TOKENS_FALLBACK + ), compact_threshold=config.get("compact_threshold", 0.92), target_usage=config.get("target_usage", 0.50), protected_recent=config.get("protected_recent", 0.30), @@ -253,7 +305,8 @@ class SimpleContextManager: def __init__( self, - max_tokens: int = 200_000, + max_tokens: int | None = None, + max_tokens_fallback: int = DEFAULT_MAX_TOKENS_FALLBACK, compact_threshold: float = 0.92, target_usage: float = 0.50, protected_recent: float = 0.30, @@ -272,7 +325,12 @@ def __init__( Initialize the context manager. Args: - max_tokens: Maximum context size in tokens + max_tokens: Cap on the effective token budget, or None for no + cap (the default: use the whole window the model allows). + Applied as min(model_derived_budget, max_tokens), so a + value above the model's own window is a no-op. + max_tokens_fallback: Budget used only when the provider reports + no usable window. Not a cap. compact_threshold: Trigger compaction at this usage ratio (0.0-1.0) target_usage: Compact down to this usage ratio (0.0-1.0) protected_recent: Always protect last N% of messages (0.0-1.0) @@ -312,6 +370,10 @@ def __init__( self.messages: list[dict[str, Any]] = [] self.max_tokens = max_tokens + self.max_tokens_fallback = max_tokens_fallback + # Last budget handed to a request; used for honest usage logging + # before any request has been made (see add_message). + self._last_effective_budget: int | None = None self.compact_threshold = compact_threshold self.target_usage = target_usage self.protected_recent = protected_recent @@ -554,7 +616,12 @@ async def add_message(self, message: dict[str, Any]) -> None: await self._emit_tool_result_ingress_truncation(truncation_event) token_count = self._estimate_tokens(self.messages) - usage = token_count / self.max_tokens + # Report usage against the budget the last request actually ran on. + # Dividing by `max_tokens` (a cap that is usually unset, and never + # the governing number when a provider publishes a window) produced + # a percentage of nothing meaningful. + denominator = self._last_effective_budget or self.max_tokens_fallback + usage = token_count / denominator if denominator else 0.0 logger.debug( f"Added message: {message.get('role', 'unknown')} - " f"{len(self.messages)} total messages, {token_count:,} tokens " @@ -731,12 +798,14 @@ async def get_messages_for_request( Args: token_budget: Optional explicit token limit (deprecated, prefer provider). provider: Optional provider instance for dynamic budget calculation. - If provided, budget = context_window - max_output_tokens - safety_margin. + If provided, budget = context_window - max_output_tokens - safety_margin, + then capped at `max_tokens` when one is configured. Returns: Messages ready for LLM request, compacted if necessary. """ budget = self._calculate_budget(token_budget, provider) + self._last_effective_budget = budget # Reserve token budget for potential compaction notice (if enabled) effective_budget = budget @@ -2352,13 +2421,41 @@ def _format_affected_items(self, level: int, stats: dict[str, Any]) -> str: ) def _calculate_budget(self, token_budget: int | None, provider: Any | None) -> int: - """Calculate effective token budget from provider or fallback to config. + """Effective token budget: what the model offers, capped by `max_tokens`. + + Two steps, deliberately separate: + + 1. DERIVE what is available (:meth:`_derive_budget`) -- the provider's + own window, or the fallback when it publishes none. + 2. CAP it at the configured `max_tokens`, if one is set. + + The cap is applied to EVERY derivation path, with no exceptions, so + the rule is one sentence: *the effective budget is the lower of what + the model allows and what you asked for.* `min()` can only lower a + budget, never raise it, which is why it is safe on every path -- a + smaller budget compacts earlier, it can never overflow a request. + + `max_tokens` defaults to None (no cap). Setting it ABOVE the model's + own window is a no-op by construction rather than an error. + """ + budget = self._derive_budget(token_budget, provider) + + if self.max_tokens is not None and budget > self.max_tokens: + logger.info( + f"Budget capped by max_tokens: {budget:,} -> {self.max_tokens:,}" + ) + return self.max_tokens + + return budget + + def _derive_budget(self, token_budget: int | None, provider: Any | None) -> int: + """Budget available BEFORE the `max_tokens` cap is applied. Priority: 1. Explicit token_budget parameter (deprecated but supported) 2. Provider model info (context_window - reserved_output - safety_margin) 3. Provider defaults (legacy: some providers may put limits here) - 4. Configured max_tokens fallback + 4. Configured max_tokens_fallback Note: We reserve only 50% of max_output_tokens since most responses are much smaller than the maximum. This prevents over-conservative budgets @@ -2415,9 +2512,12 @@ def _calculate_budget(self, token_budget: int | None, provider: Any | None) -> i except Exception as e: logger.debug(f"Could not get budget from provider: {e}") - # Fall back to configured max_tokens - logger.info(f"Using fallback max_tokens budget: {self.max_tokens:,}") - return self.max_tokens + # No provider window available anywhere -- use the calibrated fallback. + logger.info( + f"Provider published no usable window; using max_tokens_fallback: " + f"{self.max_tokens_fallback:,}" + ) + return self.max_tokens_fallback def _estimate_tokens(self, messages: list[dict[str, Any]]) -> int: """Rough token estimation (chars / 4).""" diff --git a/tests/test_compaction_trigger_provenance.py b/tests/test_compaction_trigger_provenance.py index 2cd263e..4002471 100644 --- a/tests/test_compaction_trigger_provenance.py +++ b/tests/test_compaction_trigger_provenance.py @@ -6,16 +6,18 @@ trigger = compact_threshold * effective_budget -and `effective_budget` is derived **from the provider**, not from the -configured `max_tokens`. `_calculate_budget()` only falls back to +and `effective_budget` is derived **from the provider** and then CAPPED by +the configured `max_tokens`. `_calculate_budget()` only falls back to `self.max_tokens` when no provider is passed *or* the provider exposes no usable window information -- and the orchestrator always passes a provider (`loop-streaming` calls `context.get_messages_for_request(provider=provider)`). -That makes `max_tokens` a **silently dead knob in production**: the shipped -foundation bundle sets `context.config.max_tokens: 300000`, and it has no -effect on when compaction fires. An operator who lowers or raises it to move -the trigger gets no wire effect at all. +HISTORY: `max_tokens` used to be a *fallback only* -- consulted solely when +no provider published a window -- which made it a silently dead knob in +production, since the orchestrator always passes a provider. It is now a +**cap**: `min(derived_budget, max_tokens)`, defaulting to None (no cap). +Lowering it moves the trigger; raising it above the model window is a no-op. +These tests pin BOTH halves of that contract. This is not hypothetical. The cadence probe ("PROBE 4", capture root `.amplifier/evaluation/treatment-validation/20260901-cadence/`) could not move @@ -23,7 +25,8 @@ in-container to add `budget = min(budget, self.max_tokens)` before its arms would compact at all. Its own note records why -- "the loop always passes the provider ... so the configured max_tokens is dead and compaction never fires -in a bounded run" (`scenarios/_harness/configure_cell.py`). +in a bounded run" (`scenarios/_harness/configure_cell.py`). That patch is +exactly the behavior this module now ships, as a supported option. Before this file, **no test in the suite exercised the provider-derived budget path at all** -- every existing test constructs the manager with `max_tokens` @@ -40,7 +43,11 @@ import pytest -from amplifier_module_context_simple import SimpleContextManager +from amplifier_module_context_simple import ( + DEFAULT_MAX_TOKENS_FALLBACK, + SimpleContextManager, + _validated_max_tokens, +) # --------------------------------------------------------------------------- # Values quoted from named sources, not invented here. @@ -113,36 +120,60 @@ def _expected_budget(context_window: int, max_output_tokens: int) -> int: # --------------------------------------------------------------------------- -def test_provider_model_info_budget_overrides_configured_max_tokens(): - """A provider that reports a window makes `max_tokens` irrelevant.""" - context = SimpleContextManager(max_tokens=FOUNDATION_CONFIGURED_MAX_TOKENS) +def test_provider_model_info_budget_is_used_when_no_cap_is_set(): + """With `max_tokens` unset (the default), the provider's window governs.""" + context = SimpleContextManager() provider = _ProviderWithModelInfo(context_window=1_000_000, max_output_tokens=128_000) budget = context._calculate_budget(None, provider) assert budget == _expected_budget(1_000_000, 128_000) == 931_904 - assert budget != FOUNDATION_CONFIGURED_MAX_TOKENS, ( - "The configured max_tokens must not be what sets the budget when a " - "provider is present -- if this ever becomes true, the dead-knob trap " - "documented in this file has been fixed and the README section " - "'Where the compaction trigger comes from' needs updating." + + +def test_max_tokens_caps_the_provider_budget(): + """The fix: a configured `max_tokens` LOWERS a larger provider window.""" + context = SimpleContextManager(max_tokens=FOUNDATION_CONFIGURED_MAX_TOKENS) + provider = _ProviderWithModelInfo(context_window=1_000_000, max_output_tokens=128_000) + + budget = context._calculate_budget(None, provider) + + assert budget == FOUNDATION_CONFIGURED_MAX_TOKENS, ( + "max_tokens is a cap: with a 1M-window provider and a 300k cap, the " + "effective budget must be the cap, not the window." ) -def test_provider_defaults_budget_overrides_configured_max_tokens(): +def test_max_tokens_above_the_model_window_is_a_no_op(): + """The lower of the two always wins -- a cap cannot RAISE a budget.""" + context = SimpleContextManager(max_tokens=5_000_000) + provider = _ProviderWithModelInfo(context_window=1_000_000, max_output_tokens=128_000) + + assert context._calculate_budget(None, provider) == _expected_budget( + 1_000_000, 128_000 + ) + + +def test_provider_defaults_budget_is_used_when_no_cap_is_set(): """Same, via the legacy `get_info().defaults` path.""" - context = SimpleContextManager(max_tokens=FOUNDATION_CONFIGURED_MAX_TOKENS) + context = SimpleContextManager() provider = _ProviderWithDefaultsOnly(context_window=200_000, max_output_tokens=64_000) budget = context._calculate_budget(None, provider) assert budget == _expected_budget(200_000, 64_000) == 163_904 - assert budget != FOUNDATION_CONFIGURED_MAX_TOKENS -def test_max_tokens_is_used_only_when_the_provider_reports_no_window(): - """`max_tokens` is a *fallback*, and only a fallback.""" - context = SimpleContextManager(max_tokens=FOUNDATION_CONFIGURED_MAX_TOKENS) +def test_cap_applies_to_the_legacy_defaults_path_too(): + """The cap has no exceptions -- it applies on every derivation path.""" + context = SimpleContextManager(max_tokens=100_000) + provider = _ProviderWithDefaultsOnly(context_window=200_000, max_output_tokens=64_000) + + assert context._calculate_budget(None, provider) == 100_000 + + +def test_fallback_is_used_only_when_the_provider_reports_no_window(): + """`max_tokens_fallback` -- not `max_tokens` -- answers when nothing else can.""" + context = SimpleContextManager(max_tokens_fallback=FOUNDATION_CONFIGURED_MAX_TOKENS) assert context._calculate_budget(None, None) == FOUNDATION_CONFIGURED_MAX_TOKENS assert ( @@ -151,6 +182,21 @@ def test_max_tokens_is_used_only_when_the_provider_reports_no_window(): ) +def test_default_fallback_is_the_calibrated_constant(): + """An unconfigured manager falls back to the documented 200k, not to None.""" + context = SimpleContextManager() + + assert context._calculate_budget(None, None) == DEFAULT_MAX_TOKENS_FALLBACK + assert DEFAULT_MAX_TOKENS_FALLBACK == 200_000 + + +def test_cap_applies_to_the_fallback_path_too(): + """A cap below the fallback still wins -- one rule, no exceptions.""" + context = SimpleContextManager(max_tokens=50_000, max_tokens_fallback=300_000) + + assert context._calculate_budget(None, _ProviderWithNoWindowInfo()) == 50_000 + + def test_explicit_token_budget_still_wins_over_everything(): """Priority 1 is unchanged: an explicit budget short-circuits the rest.""" context = SimpleContextManager(max_tokens=FOUNDATION_CONFIGURED_MAX_TOKENS) @@ -174,27 +220,60 @@ async def _fill(context: SimpleContextManager, pairs: int = 20, chars: int = 5_0 @pytest.mark.asyncio -async def test_lowering_max_tokens_does_not_move_the_trigger_when_a_provider_is_present(): - """THE TRAP, demonstrated. - - Two managers configured with the two cadence-harness forcing values - (45,000 and 70,000) see the *same* provider and the *same* history, and - neither compacts -- because the provider's budget, not `max_tokens`, is - what the threshold is applied to. The config knob that the cadence probe - "moved" has no effect through the shipped code path. +async def test_lowering_max_tokens_DOES_move_the_trigger_when_a_provider_is_present(): + """THE TRAP, INVERTED -- this is the behavior change. + + Before the cap existed, both cadence-harness forcing values (45,000 and + 70,000) were dead: a provider reporting a 200,000-token window produced a + 163,904 budget and ~50,000 tokens of history never crossed the threshold, + no matter what `max_tokens` said. The harness had to patch module source + to get the effect this test now asserts as shipped behavior. + + The two harness values now land on OPPOSITE sides of the same ~50,000-token + history, which is the sharpest available proof that the knob is live: 45,000 + caps low enough to compact (0.92 * 45,000 = 41,400 < ~50,000), 70,000 does + not (0.92 * 70,000 = 64,400 > ~50,000). Before the cap, both were dead and + neither compacted. """ provider = _ProviderWithModelInfo(context_window=200_000, max_output_tokens=64_000) - for configured in (CAD_TODAY_MAX_TOKENS, CAD_FEWER_MAX_TOKENS): - context = SimpleContextManager(max_tokens=configured) - await _fill(context) - await context.get_messages_for_request(provider=provider) + low = SimpleContextManager(max_tokens=CAD_TODAY_MAX_TOKENS) + await _fill(low) + await low.get_messages_for_request(provider=provider) + assert low._last_compaction_stats is not None, ( + f"max_tokens={CAD_TODAY_MAX_TOKENS:,} must cap the 163,904 provider " + f"budget and move the trigger; ~50,000 tokens of history is over " + f"compact_threshold * {CAD_TODAY_MAX_TOKENS:,}." + ) - assert context._last_compaction_stats is None, ( - f"max_tokens={configured:,} must not move the compaction trigger " - f"while a provider reporting a 200,000-token window is present; " - f"the effective budget is {_expected_budget(200_000, 64_000):,}." - ) + high = SimpleContextManager(max_tokens=CAD_FEWER_MAX_TOKENS) + await _fill(high) + await high.get_messages_for_request(provider=provider) + assert high._last_compaction_stats is None, ( + f"max_tokens={CAD_FEWER_MAX_TOKENS:,} caps the budget but stays above " + f"this history -- raising the cap must move the trigger LATER, which is " + f"what makes the knob a real dial rather than an on/off switch." + ) + + +@pytest.mark.asyncio +async def test_an_unset_cap_leaves_the_provider_budget_alone(): + """The no-cap default is the old behavior, exactly. + + Same provider, same history as the test above, with `max_tokens` unset: + the 163,904 provider budget governs and ~50,000 tokens does not reach it. + This is what makes the cap opt-in rather than a silent clamp. + """ + provider = _ProviderWithModelInfo(context_window=200_000, max_output_tokens=64_000) + context = SimpleContextManager() + await _fill(context) + await context.get_messages_for_request(provider=provider) + + assert context._last_compaction_stats is None, ( + "With no cap configured, the effective budget must remain the " + f"provider's {_expected_budget(200_000, 64_000):,} and this history " + "must not compact." + ) @pytest.mark.asyncio @@ -205,14 +284,14 @@ async def test_the_same_history_and_config_does_compact_once_the_provider_is_gon becomes live, so compaction fires. The only difference between "trigger dead" and "trigger live" is whether a provider was passed. """ - context = SimpleContextManager(max_tokens=CAD_TODAY_MAX_TOKENS) + context = SimpleContextManager(max_tokens_fallback=CAD_TODAY_MAX_TOKENS) await _fill(context) await context.get_messages_for_request() # no provider -> fallback budget stats = context._last_compaction_stats assert stats is not None, ( - "With no provider, max_tokens is the budget and this history is well " - "over compact_threshold * max_tokens -- compaction must fire." + "With no provider, max_tokens_fallback is the budget and this history " + "is well over compact_threshold * fallback -- compaction must fire." ) assert stats["after_tokens"] < stats["before_tokens"] @@ -274,3 +353,30 @@ def test_capping_the_budget_at_cad_fewer_value_would_compact_EARLIER_not_later( f"trigger from {shipped_trigger:,.0f} tokens to {capped_trigger:,.0f} " "-- earlier, not later." ) + + +# --------------------------------------------------------------------------- +# 4. A nonsense cap is refused loudly, not honored. +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize("bad", [0, -1, -500_000, "500000", 1.5, True, False]) +def test_invalid_max_tokens_is_ignored_rather_than_honored(bad): + """A cap of 0 or a non-int would drive the budget to <= 0. + + `_should_compact`'s `budget > 0` guard would then force usage to 0 and + silently disable compaction entirely -- the exact opposite of what someone + setting a cap is asking for. So an invalid value degrades to "no cap", + with a warning, matching how `compaction_notice_token_reserve` handles a + self-defeating value. + """ + assert _validated_max_tokens(bad) is None + + +@pytest.mark.parametrize("good", [1, 200_000, 500_000, 5_000_000]) +def test_valid_max_tokens_passes_through(good): + assert _validated_max_tokens(good) == good + + +def test_none_means_no_cap(): + assert _validated_max_tokens(None) is None From cd79245614ac3c581c69418fd289e3eeb88e7827 Mon Sep 17 00:00:00 2001 From: Amplifier <240397093+microsoft-amplifier@users.noreply.github.com> Date: Tue, 15 Sep 2026 07:38:49 -0700 Subject: [PATCH 2/2] feat: emit context:budget so the effective budget is an artifact, not a log line DTU validation of the max_tokens cap could not answer "what budget did this session actually run on?" after the fact. _derive_budget logs the number it picks, but the log never reaches the session record and did not surface on stdout even at AMPLIFIER_LOG_LEVEL=INFO. That is the first question anyone asks when compaction fires earlier than expected or a cap is suspected, and it was unanswerable from the record. Emits `context:budget` carrying the effective budget and its provenance: source (provider_model_info | provider_defaults | max_tokens_fallback | explicit), the window and reserve it was derived from, derived_budget, max_tokens, max_tokens_fallback, capped, effective_budget. Two properties matter as much as the number: - Fires at the DELIVERY boundary, not at calculation time. A request that raises or is cancelled rolls back and must leave no observable trace -- an invariant test_request_retention pins explicitly, and which the first version of this change broke: emitting on calculation made three rollback tests fail. Both delivery returns emit, including the common no-compaction path. - Fires on CHANGE, not per request. The value is stable for most of a session, so per-request emission would drown the record it exists to make readable; a change (provider swap, mid-session cap, fallback kicking in) is always worth a line. Observability is never load-bearing: no hooks, or an emitter that raises, is logged and swallowed -- the same contract as context:compaction and _emit_tool_result_ingress_truncation, both covered by tests here. Verified against the real GeminiProvider: uncapped reports effective=1,011,712 source=provider_defaults capped=False; with max_tokens=500000 it reports effective=500,000 capped=True. Tests: 7 new in test_budget_event.py. test_request_retention's exact-event-list assertion updated to include the new event, with a note on why exactly one appears. 173 passed, 1 xfailed. Generated with Amplifier Co-Authored-By: Amplifier <240397093+microsoft-amplifier@users.noreply.github.com> --- README.md | 22 +++ amplifier_module_context_simple/__init__.py | 77 ++++++++++- tests/test_budget_event.py | 144 ++++++++++++++++++++ tests/test_request_retention.py | 9 +- 4 files changed, 246 insertions(+), 6 deletions(-) create mode 100644 tests/test_budget_event.py diff --git a/README.md b/README.md index 819a7b0..c83a3b8 100644 --- a/README.md +++ b/README.md @@ -216,6 +216,28 @@ overrides: max_tokens_fallback: 400000 ``` +### Seeing the budget a session actually ran on + +`context-simple` emits a `context:budget` event carrying the effective budget +and where it came from: + +```json +{"source": "provider_defaults", "context_window": 1048576, + "max_output_tokens": 65536, "reserved_output": 32768, + "derived_budget": 1011712, "max_tokens": 500000, + "max_tokens_fallback": 200000, "capped": true, "effective_budget": 500000} +``` + +`source` is one of `provider_model_info`, `provider_defaults`, +`max_tokens_fallback` or `explicit`, and `capped` says whether `max_tokens` +bit. The module also logs the same number, but a log line is not an artifact -- +it never reaches the session record, which made "what budget did this session +run on?" unanswerable after the fact. + +It fires at the delivery boundary, so a cancelled or failed request leaves no +trace, and only when the value CHANGES, so a long session carries a readable +handful of lines rather than one per request. + ### History: this knob used to be dead Before the cap existed, `max_tokens` was consulted **only** at step 1.4 -- diff --git a/amplifier_module_context_simple/__init__.py b/amplifier_module_context_simple/__init__.py index d96a193..d4cc625 100644 --- a/amplifier_module_context_simple/__init__.py +++ b/amplifier_module_context_simple/__init__.py @@ -374,6 +374,12 @@ def __init__( # Last budget handed to a request; used for honest usage logging # before any request has been made (see add_message). self._last_effective_budget: int | None = None + # Provenance of the most recent budget calculation, and the last + # payload emitted, so `context:budget` fires on CHANGE rather than + # once per request (a per-request event would drown the record it + # is meant to make readable). + self._budget_provenance: dict[str, Any] = {} + self._last_emitted_budget: dict[str, Any] | None = None self.compact_threshold = compact_threshold self.target_usage = target_usage self.protected_recent = protected_recent @@ -576,6 +582,42 @@ async def _emit_tool_result_ingress_truncation(self, event_data: dict[str, Any]) event_data["max_tool_result_bytes"], ) + async def _emit_budget(self) -> None: + """Publish the effective budget where a session record can see it. + + `_derive_budget` logs its result, but a log line is not an artifact: + it never reaches the session record and does not survive the process. + So "what budget did this session actually run on?" -- the first + question anyone asks when compaction fires early or a cap is + suspected -- was unanswerable after the fact. + + Emitted on CHANGE, not per request: the value is stable for most of a + session, so a per-request event would be noise, while a change is + always worth a line (a provider swap, a mid-session cap, a fallback + kicking in). + + Emitted at the DELIVERY boundary, not when the budget is calculated. A + request that raises or is cancelled rolls back and must leave no + observable trace -- an invariant `test_request_retention` pins + explicitly -- so this fires only once a view is actually being + returned, the same boundary `context:compaction` commits at. + + Observability must never break a request, so a failed emission is + logged and swallowed -- the same contract as + `_emit_tool_result_ingress_truncation` and `context:compaction`. + """ + payload = dict(self._budget_provenance) + if not payload or payload == self._last_emitted_budget: + return + + self._last_emitted_budget = payload + if self._hooks is None: + return + try: + await self._hooks.emit("context:budget", payload) + except Exception as e: + logger.warning(f"Could not emit budget event: {e}") + async def add_message(self, message: dict[str, Any]) -> None: """Add a message to the context. @@ -929,6 +971,7 @@ async def get_messages_for_request( compacted = sticky_view else: self._check_retained_budget(working_messages, budget) + await self._emit_budget() return self._strip_internal_metadata(working_messages) # Append compaction notice at the TAIL if enabled and level threshold met. @@ -989,6 +1032,7 @@ async def get_messages_for_request( await self._hooks.emit("context:compaction", deferred_hard_fit_stats) except Exception as e: logger.warning(f"Could not emit compaction event: {e}") + await self._emit_budget() return self._strip_internal_metadata(compacted) # Metadata keys that are internal bookkeeping only and must never cross @@ -2438,15 +2482,24 @@ def _calculate_budget(self, token_budget: int | None, provider: Any | None) -> i `max_tokens` defaults to None (no cap). Setting it ABOVE the model's own window is a no-op by construction rather than an error. """ - budget = self._derive_budget(token_budget, provider) + self._budget_provenance = {"source": "unknown"} + derived = self._derive_budget(token_budget, provider) - if self.max_tokens is not None and budget > self.max_tokens: + budget_capped = self.max_tokens is not None and derived > self.max_tokens + if budget_capped: logger.info( - f"Budget capped by max_tokens: {budget:,} -> {self.max_tokens:,}" + f"Budget capped by max_tokens: {derived:,} -> {self.max_tokens:,}" ) - return self.max_tokens - return budget + self._budget_provenance = { + **self._budget_provenance, + "derived_budget": derived, + "max_tokens": self.max_tokens, + "max_tokens_fallback": self.max_tokens_fallback, + "capped": budget_capped, + "effective_budget": self.max_tokens if budget_capped else derived, + } + return self.max_tokens if budget_capped else derived def _derive_budget(self, token_budget: int | None, provider: Any | None) -> int: """Budget available BEFORE the `max_tokens` cap is applied. @@ -2464,6 +2517,7 @@ def _derive_budget(self, token_budget: int | None, provider: Any | None) -> int: # Explicit budget takes precedence (for backward compatibility) if token_budget is not None: logger.debug(f"Using explicit token_budget: {token_budget}") + self._budget_provenance = {"source": "explicit"} return token_budget safety_margin = 4096 # Buffer to avoid hitting hard limits @@ -2487,6 +2541,12 @@ def _derive_budget(self, token_budget: int | None, provider: Any | None) -> int: f"(context={context_window:,}, reserved_output={reserved_output:,} " f"[{output_reserve_fraction:.0%} of {max_output:,}])" ) + self._budget_provenance = { + "source": "provider_model_info", + "context_window": context_window, + "max_output_tokens": max_output, + "reserved_output": reserved_output, + } return budget # Check provider info defaults (legacy approach) @@ -2503,6 +2563,12 @@ def _derive_budget(self, token_budget: int | None, provider: Any | None) -> int: f"(context={context_window:,}, reserved_output={reserved_output:,} " f"[{output_reserve_fraction:.0%} of {max_output_tokens:,}])" ) + self._budget_provenance = { + "source": "provider_defaults", + "context_window": context_window, + "max_output_tokens": max_output_tokens, + "reserved_output": reserved_output, + } return budget else: logger.debug( @@ -2517,6 +2583,7 @@ def _derive_budget(self, token_budget: int | None, provider: Any | None) -> int: f"Provider published no usable window; using max_tokens_fallback: " f"{self.max_tokens_fallback:,}" ) + self._budget_provenance = {"source": "max_tokens_fallback"} return self.max_tokens_fallback def _estimate_tokens(self, messages: list[dict[str, Any]]) -> int: diff --git a/tests/test_budget_event.py b/tests/test_budget_event.py new file mode 100644 index 0000000..20ef8de --- /dev/null +++ b/tests/test_budget_event.py @@ -0,0 +1,144 @@ +"""`context:budget` -- the effective budget, as an artifact rather than a log line. + +Why this file exists +-------------------- +`_derive_budget` logs the number it picks, but a log line is not an artifact: +DTU validation of the max_tokens-cap change could not answer "what budget did +this session actually run on?" after the fact. The log never reached the +session record and did not surface on stdout even at INFO. + +That is the first question anyone asks when compaction fires earlier than +expected, or when a cap is suspected but not confirmed. This event makes it +answerable from the record. + +Two properties matter as much as the number itself: + +* it fires at the DELIVERY boundary, so a cancelled or failed request leaves + no trace (the rollback invariant `test_request_retention` pins), and +* it fires on CHANGE, so a long session carries a readable handful of lines + rather than one per request. +""" + +from __future__ import annotations + +import pytest + +from amplifier_module_context_simple import SimpleContextManager + + +class _Hooks: + def __init__(self) -> None: + self.events: list[tuple[str, dict]] = [] + + async def emit(self, event: str, data: dict) -> None: + self.events.append((event, data)) + + +class _Provider: + def __init__(self, context_window: int, max_output_tokens: int) -> None: + self._defaults = { + "context_window": context_window, + "max_output_tokens": max_output_tokens, + } + + def get_info(self): + return type("_Info", (), {"defaults": self._defaults})() + + +def _budget_events(hooks: _Hooks) -> list[dict]: + return [data for event, data in hooks.events if event == "context:budget"] + + +@pytest.mark.asyncio +async def test_budget_is_emitted_with_its_provenance(): + hooks = _Hooks() + context = SimpleContextManager(hooks=hooks) + await context.add_message({"role": "user", "content": "hi"}) + + await context.get_messages_for_request(provider=_Provider(1_048_576, 65_536)) + + (event,) = _budget_events(hooks) + assert event["effective_budget"] == 1_011_712 + assert event["source"] == "provider_defaults" + assert event["context_window"] == 1_048_576 + assert event["capped"] is False + + +@pytest.mark.asyncio +async def test_a_cap_is_visible_in_the_event(): + """The question this exists to answer: was I capped, and by what?""" + hooks = _Hooks() + context = SimpleContextManager(max_tokens=500_000, hooks=hooks) + await context.add_message({"role": "user", "content": "hi"}) + + await context.get_messages_for_request(provider=_Provider(1_048_576, 65_536)) + + (event,) = _budget_events(hooks) + assert event["capped"] is True + assert event["effective_budget"] == 500_000 + assert event["derived_budget"] == 1_011_712 + assert event["max_tokens"] == 500_000 + + +@pytest.mark.asyncio +async def test_the_fallback_path_names_itself(): + hooks = _Hooks() + context = SimpleContextManager(hooks=hooks) + await context.add_message({"role": "user", "content": "hi"}) + + await context.get_messages_for_request() + + (event,) = _budget_events(hooks) + assert event["source"] == "max_tokens_fallback" + assert event["effective_budget"] == 200_000 + + +@pytest.mark.asyncio +async def test_it_fires_on_change_not_per_request(): + hooks = _Hooks() + context = SimpleContextManager(hooks=hooks) + provider = _Provider(1_048_576, 65_536) + await context.add_message({"role": "user", "content": "hi"}) + + for _ in range(5): + await context.get_messages_for_request(provider=provider) + + assert len(_budget_events(hooks)) == 1 + + +@pytest.mark.asyncio +async def test_a_changed_budget_emits_again(): + hooks = _Hooks() + context = SimpleContextManager(hooks=hooks) + await context.add_message({"role": "user", "content": "hi"}) + + await context.get_messages_for_request(provider=_Provider(1_048_576, 65_536)) + await context.get_messages_for_request(provider=_Provider(200_000, 64_000)) + + events = _budget_events(hooks) + assert [e["effective_budget"] for e in events] == [1_011_712, 163_904] + + +@pytest.mark.asyncio +async def test_no_hooks_is_not_an_error(): + """Observability must never be load-bearing.""" + context = SimpleContextManager() + await context.add_message({"role": "user", "content": "hi"}) + + result = await context.get_messages_for_request(provider=_Provider(200_000, 64_000)) + + assert result + + +@pytest.mark.asyncio +async def test_a_failing_emitter_does_not_break_the_request(): + class _Broken(_Hooks): + async def emit(self, event: str, data: dict) -> None: + raise RuntimeError("hook exploded") + + context = SimpleContextManager(hooks=_Broken()) + await context.add_message({"role": "user", "content": "hi"}) + + result = await context.get_messages_for_request(provider=_Provider(200_000, 64_000)) + + assert result diff --git a/tests/test_request_retention.py b/tests/test_request_retention.py index 09714e3..b3be8fc 100644 --- a/tests/test_request_retention.py +++ b/tests/test_request_retention.py @@ -329,7 +329,14 @@ async def emit(self, event: str, data: dict) -> None: ordinary = await context.get_messages_for_request() assert ordinary == await context.get_messages_for_request() assert len(ordinary) < len(canonical) - assert [event for event, _ in emitted] == ["context:compaction"] + # The cancelled delivery emitted no further compaction event. `context:budget` + # appears exactly once here, from the first ordinary request that reached the + # delivery boundary -- it fires on CHANGE, so the identical second request + # adds nothing. + assert [event for event, _ in emitted] == [ + "context:compaction", + "context:budget", + ] @pytest.mark.asyncio