diff --git a/config/config.yaml b/config/config.yaml index f4ff13a..2c7279a 100644 --- a/config/config.yaml +++ b/config/config.yaml @@ -282,6 +282,17 @@ pinch: budget_tokens: 50000 keep_last_turns: 4 max_summarize_chars: 4000 + # keep_last_turns has no size limit inside it: an entire autonomous + # tool-call loop with no new user message can be one protected turn, and + # one outsized tool result inside it (a full verbose test run, a huge file + # read) ships verbatim regardless of size. Measured live 2026-09-06: a + # 324k-token conversation shrank only ~8% because nearly all of it sat + # inside the protected window. This closes that gap: any tool result + # inside the protected window over this many characters still gets the + # same head/tail elision candidates get. Deliberately a much higher bar + # than max_summarize_chars -- recent results are more likely to still + # matter -- so it only catches true outliers. Set to null to disable. + protected_max_chars: 20000 relevance: # Off by default, matching every other new-and-unproven knob in this # project — and specifically requires pinch.enabled too, since this has no diff --git a/docs/pinch.md b/docs/pinch.md index 975cb8e..4b810ab 100644 --- a/docs/pinch.md +++ b/docs/pinch.md @@ -99,6 +99,7 @@ pinch: budget_tokens: 50000 keep_last_turns: 4 max_summarize_chars: 4000 + protected_max_chars: 20000 relevance: enabled: true model: "nomic-embed-text" @@ -113,6 +114,7 @@ pinch: | `pinch.budget_tokens` | `50000` | Above this estimated token count the conversation is pruned. Must be `> 0`. | | `pinch.keep_last_turns` | `4` | How many recent user turns (plus their assistant replies and tool results) are protected from pruning. Must be `> 0`. | | `pinch.max_summarize_chars` | `4000` | Tool results longer than this many chars are summarized in place; shorter ones are collapsed to a placeholder. Must be `>= 3000` — below that summarization would grow the message. | +| `pinch.protected_max_chars` | `20000` | `keep_last_turns` has no size limit inside it — an entire autonomous tool-call loop with no new user message can be one protected turn, so one outsized tool result inside it (a full verbose test run, a huge file read) used to ship verbatim regardless of size. Measured live 2026-09-06: a 324k-token conversation shrank only ~8% because nearly all of it sat inside the protected window. Any tool result inside the protected window over this many chars now gets the same head/tail elision candidates get — deliberately a much higher bar than `max_summarize_chars`, since recent results are more likely to still matter, so it only catches true outliers. Must be `>= 3000`, or `null` to disable. | | `pinch.relevance.enabled` | `true` | Embedding-model relevance scoring. Requires `pinch.enabled` too. Any failure reverts to uniform trimming. | | `pinch.relevance.model` | `nomic-embed-text` | An **embedding** model — never `classifier.model` or `verification.model`. | | `pinch.relevance.base_url` | `http://localhost:11434/v1` | OpenAI-compatible embeddings endpoint on the same local Ollama. | diff --git a/src/config.py b/src/config.py index 94be07c..9694199 100644 --- a/src/config.py +++ b/src/config.py @@ -575,6 +575,15 @@ class PinchConfig(StrictModel): keep_last_turns: int = 4 # Tool results longer than this many characters are summarized in place. max_summarize_chars: int = 4000 + # keep_last_turns has no size limit inside it -- an entire autonomous + # tool-call loop with no new user message can be one protected turn, and + # one outsized tool result inside it (a full verbose test run, a huge + # file read) ships verbatim regardless of size. Measured live: a + # 324k-token conversation shrank only ~8% because nearly all of it sat + # inside the protected window. This is deliberately a much higher bar + # than max_summarize_chars -- recent results are more likely to still + # matter -- so it only catches true outliers. None disables it. + protected_max_chars: Optional[int] = 20000 relevance: PinchRelevanceConfig = PinchRelevanceConfig() @field_validator("budget_tokens") @@ -601,6 +610,17 @@ class PinchConfig(StrictModel): ) return v + @field_validator("protected_max_chars") + @classmethod + def protected_max_chars_valid(cls, v: Optional[int]) -> Optional[int]: + if v is not None and v < 3000: + raise ValueError( + "pinch.protected_max_chars must be >= 3000, or null to " + "disable (below this, elision grows the message, same " + "floor as max_summarize_chars)" + ) + return v + class SessionCacheConfig(StrictModel): """Per-session classification cache (in-memory, process lifetime). diff --git a/src/context_prune.py b/src/context_prune.py index 65eaa0a..4a8ec8a 100644 --- a/src/context_prune.py +++ b/src/context_prune.py @@ -243,6 +243,7 @@ def prune_context( max_summarize_chars: int = PinchConfig.model_fields["max_summarize_chars"].default, relevance_order: Optional[list[int]] = None, extra_fixed_tokens: int = 0, + protected_max_chars: Optional[int] = None, ) -> tuple[list[dict], dict]: """Trim old tool results once a conversation exceeds ``budget_tokens``. @@ -260,6 +261,22 @@ def prune_context( Only runs (and only mutates anything) when the estimate actually exceeds the budget; otherwise the original list is returned untouched. + + ``protected_max_chars`` closes a real gap in the ``keep_last_turns`` + guarantee: that window protects everything from the last N *user-turn* + boundaries onward, with no size limit inside it. In a long autonomous + tool-call loop (no new user message between calls), an entire multi-hour + session can be a single protected turn, and one outsized tool result + inside it — a full verbose test-suite run, a huge file read — ships + verbatim no matter how large. Measured live: a 324k-token conversation + shrank only ~8% because nearly all of it sat inside the protected window. + When set, any tool result inside the protected window whose text exceeds + this many characters still gets the same head/tail elision candidates get + -- deliberately a much higher bar than ``max_summarize_chars`` (recent + results are more likely to matter), so this only catches true outliers, + never ordinary recent tool output. ``None`` (the pure function's default) + preserves the exact historical behavior; ``PinchConfig`` supplies a real + default so production gets the fix without every caller needing to opt in. """ orig_tokens = sum(estimate_tokens(extract_text(m)) for m in messages) + extra_fixed_tokens if orig_tokens <= budget_tokens: @@ -387,6 +404,31 @@ def prune_context( else: pruned.append(msg) + if protected_max_chars is not None: + for i in range(protected_from, len(pruned)): + msg = pruned[i] + if msg.get("role") != "tool": + continue + content = msg.get("content") + if isinstance(content, str): + text = content + prose = content + else: + text = extract_text(msg) + prose = _text_only(msg) + if len(text) <= protected_max_chars: + continue + head_len = 1500 + tail_len = 1500 + head = prose[:head_len] + tail = prose[-tail_len:] + trimmed = len(prose) - head_len - tail_len + marker = f"\n\n[{trimmed:,} chars trimmed...]\n\n" if trimmed > 0 else "" + elided = f"{head}{marker}{tail}" + if trimmed > 0 and len(elided) < len(prose): + summarized += 1 + pruned[i] = _with_text(msg, elided) + final_tokens = sum(estimate_tokens(extract_text(m)) for m in pruned) return pruned, { "pruned": True, diff --git a/src/dispatcher.py b/src/dispatcher.py index ca43459..0ccfe4d 100644 --- a/src/dispatcher.py +++ b/src/dispatcher.py @@ -3774,6 +3774,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks): extra_fixed_tokens=_tools_overhead, ), extra_fixed_tokens=_tools_overhead, + protected_max_chars=cfg.pinch.protected_max_chars, ) if pinch_stats.get("pruned"): logs.info( @@ -4017,6 +4018,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks): extra_fixed_tokens=_tools_overhead, ), extra_fixed_tokens=_tools_overhead, + protected_max_chars=cfg.pinch.protected_max_chars, ) if pinch_stats.get("pruned"): logs.info( diff --git a/tests/test_config_endpoints.py b/tests/test_config_endpoints.py index 1accf8b..62ef7d7 100644 --- a/tests/test_config_endpoints.py +++ b/tests/test_config_endpoints.py @@ -415,6 +415,32 @@ def test_pinch_max_summarize_chars_accepts_default(raw): assert loaded.pinch.max_summarize_chars == 4000 +def test_nonminimum_pinch_protected_max_chars_is_rejected(raw): + cfg = copy.deepcopy(raw) + cfg["pinch"]["protected_max_chars"] = 100 + with pytest.raises(ValueError, match="protected_max_chars.*>=.*3000"): + RouterConfig(**cfg) + + +def test_pinch_protected_max_chars_at_valid_min_loads(raw): + cfg = copy.deepcopy(raw) + cfg["pinch"]["protected_max_chars"] = 3000 + loaded = RouterConfig(**cfg) + assert loaded.pinch.protected_max_chars == 3000 + + +def test_pinch_protected_max_chars_accepts_default(raw): + loaded = RouterConfig(**raw) + assert loaded.pinch.protected_max_chars == 20000 + + +def test_pinch_protected_max_chars_null_disables(raw): + cfg = copy.deepcopy(raw) + cfg["pinch"]["protected_max_chars"] = None + loaded = RouterConfig(**cfg) + assert loaded.pinch.protected_max_chars is None + + def test_pinch_relevance_defaults_load(raw): loaded = RouterConfig(**raw) assert loaded.pinch.relevance.enabled is True diff --git a/tests/test_context_prune.py b/tests/test_context_prune.py index 189e044..2fa6389 100644 --- a/tests/test_context_prune.py +++ b/tests/test_context_prune.py @@ -93,6 +93,63 @@ def test_recent_turn_is_protected_from_pruning(): assert "chars trimmed" in old_tool["content"] +def test_protected_max_chars_caps_outsized_recent_tool_result(): + # A protected (recent) tool result far larger than protected_max_chars + # -- e.g. a full verbose test-suite run -- still gets elided, closing + # the gap keep_last_turns otherwise leaves: no size limit inside the + # protected window at all. + messages = [ + _user("first"), + _tool("search", "old result " * 3000), + _assistant("first answer"), + _user("second"), + _tool("pytest", "PASSED " * 50000), # huge, but inside the protected window + _assistant("second answer"), + ] + out, stats = prune_context( + messages, budget_tokens=100, keep_last_turns=1, protected_max_chars=2000, + ) + recent_tool = [m for m in out if m.get("role") == "tool"][-1] + assert "chars trimmed" in recent_tool["content"] + assert len(recent_tool["content"]) < len(messages[4]["content"]) + assert stats["tokens_saved"] > 0 + + +def test_protected_max_chars_none_preserves_historical_behavior(): + # The pure function's default (None) must reproduce the exact historical + # behavior new tests weren't written against -- a huge recent tool result + # ships verbatim when the cap isn't configured. + messages = [ + _user("first"), + _tool("search", "old result " * 3000), + _assistant("first answer"), + _user("second"), + _tool("pytest", "PASSED " * 50000), + _assistant("second answer"), + ] + out, _ = prune_context(messages, budget_tokens=100, keep_last_turns=1) + recent_tool = [m for m in out if m.get("role") == "tool"][-1] + assert recent_tool["content"] == messages[4]["content"] + + +def test_protected_max_chars_does_not_touch_small_recent_results(): + # A recent tool result under the cap is left alone even when the cap is + # configured -- this only catches true outliers, never ordinary output. + messages = [ + _user("first"), + _tool("search", "old result " * 3000), + _assistant("first answer"), + _user("second"), + _tool("search", "small recent result"), + _assistant("second answer"), + ] + out, _ = prune_context( + messages, budget_tokens=100, keep_last_turns=1, protected_max_chars=2000, + ) + recent_tool = [m for m in out if m.get("role") == "tool"][-1] + assert recent_tool["content"] == "small recent result" + + def test_tool_result_role_pairing_preserved(): # The tool message keeps its name and position, so the API still parses. messages = [