fix(pinch): cap outsized tool results even inside the protected window #44

Merged
alee merged 1 commits from fix/pinch-protected-window-size-cap into main 2026-09-06 23:11:09 +00:00
7 changed files with 160 additions and 0 deletions

View File

@@ -282,6 +282,17 @@ pinch:
budget_tokens: 50000
keep_last_turns: 4
max_summarize_chars: 4000
# keep_last_turns has no size limit inside it: an entire autonomous
# tool-call loop with no new user message can be one protected turn, and
# one outsized tool result inside it (a full verbose test run, a huge file
# read) ships verbatim regardless of size. Measured live 2026-09-06: a
# 324k-token conversation shrank only ~8% because nearly all of it sat
# inside the protected window. This closes that gap: any tool result
# inside the protected window over this many characters still gets the
# same head/tail elision candidates get. Deliberately a much higher bar
# than max_summarize_chars -- recent results are more likely to still
# matter -- so it only catches true outliers. Set to null to disable.
protected_max_chars: 20000
relevance:
# Off by default, matching every other new-and-unproven knob in this
# project — and specifically requires pinch.enabled too, since this has no

View File

@@ -99,6 +99,7 @@ pinch:
budget_tokens: 50000
keep_last_turns: 4
max_summarize_chars: 4000
protected_max_chars: 20000
relevance:
enabled: true
model: "nomic-embed-text"
@@ -113,6 +114,7 @@ pinch:
| `pinch.budget_tokens` | `50000` | Above this estimated token count the conversation is pruned. Must be `> 0`. |
| `pinch.keep_last_turns` | `4` | How many recent user turns (plus their assistant replies and tool results) are protected from pruning. Must be `> 0`. |
| `pinch.max_summarize_chars` | `4000` | Tool results longer than this many chars are summarized in place; shorter ones are collapsed to a placeholder. Must be `>= 3000` — below that summarization would grow the message. |
| `pinch.protected_max_chars` | `20000` | `keep_last_turns` has no size limit inside it — an entire autonomous tool-call loop with no new user message can be one protected turn, so one outsized tool result inside it (a full verbose test run, a huge file read) used to ship verbatim regardless of size. Measured live 2026-09-06: a 324k-token conversation shrank only ~8% because nearly all of it sat inside the protected window. Any tool result inside the protected window over this many chars now gets the same head/tail elision candidates get — deliberately a much higher bar than `max_summarize_chars`, since recent results are more likely to still matter, so it only catches true outliers. Must be `>= 3000`, or `null` to disable. |
| `pinch.relevance.enabled` | `true` | Embedding-model relevance scoring. Requires `pinch.enabled` too. Any failure reverts to uniform trimming. |
| `pinch.relevance.model` | `nomic-embed-text` | An **embedding** model — never `classifier.model` or `verification.model`. |
| `pinch.relevance.base_url` | `http://localhost:11434/v1` | OpenAI-compatible embeddings endpoint on the same local Ollama. |

View File

@@ -575,6 +575,15 @@ class PinchConfig(StrictModel):
keep_last_turns: int = 4
# Tool results longer than this many characters are summarized in place.
max_summarize_chars: int = 4000
# keep_last_turns has no size limit inside it -- an entire autonomous
# tool-call loop with no new user message can be one protected turn, and
# one outsized tool result inside it (a full verbose test run, a huge
# file read) ships verbatim regardless of size. Measured live: a
# 324k-token conversation shrank only ~8% because nearly all of it sat
# inside the protected window. This is deliberately a much higher bar
# than max_summarize_chars -- recent results are more likely to still
# matter -- so it only catches true outliers. None disables it.
protected_max_chars: Optional[int] = 20000
relevance: PinchRelevanceConfig = PinchRelevanceConfig()
@field_validator("budget_tokens")
@@ -601,6 +610,17 @@ class PinchConfig(StrictModel):
)
return v
@field_validator("protected_max_chars")
@classmethod
def protected_max_chars_valid(cls, v: Optional[int]) -> Optional[int]:
if v is not None and v < 3000:
raise ValueError(
"pinch.protected_max_chars must be >= 3000, or null to "
"disable (below this, elision grows the message, same "
"floor as max_summarize_chars)"
)
return v
class SessionCacheConfig(StrictModel):
"""Per-session classification cache (in-memory, process lifetime).

View File

@@ -243,6 +243,7 @@ def prune_context(
max_summarize_chars: int = PinchConfig.model_fields["max_summarize_chars"].default,
relevance_order: Optional[list[int]] = None,
extra_fixed_tokens: int = 0,
protected_max_chars: Optional[int] = None,
) -> tuple[list[dict], dict]:
"""Trim old tool results once a conversation exceeds ``budget_tokens``.
@@ -260,6 +261,22 @@ def prune_context(
Only runs (and only mutates anything) when the estimate actually exceeds
the budget; otherwise the original list is returned untouched.
``protected_max_chars`` closes a real gap in the ``keep_last_turns``
guarantee: that window protects everything from the last N *user-turn*
boundaries onward, with no size limit inside it. In a long autonomous
tool-call loop (no new user message between calls), an entire multi-hour
session can be a single protected turn, and one outsized tool result
inside it — a full verbose test-suite run, a huge file read — ships
verbatim no matter how large. Measured live: a 324k-token conversation
shrank only ~8% because nearly all of it sat inside the protected window.
When set, any tool result inside the protected window whose text exceeds
this many characters still gets the same head/tail elision candidates get
-- deliberately a much higher bar than ``max_summarize_chars`` (recent
results are more likely to matter), so this only catches true outliers,
never ordinary recent tool output. ``None`` (the pure function's default)
preserves the exact historical behavior; ``PinchConfig`` supplies a real
default so production gets the fix without every caller needing to opt in.
"""
orig_tokens = sum(estimate_tokens(extract_text(m)) for m in messages) + extra_fixed_tokens
if orig_tokens <= budget_tokens:
@@ -387,6 +404,31 @@ def prune_context(
else:
pruned.append(msg)
if protected_max_chars is not None:
for i in range(protected_from, len(pruned)):
msg = pruned[i]
if msg.get("role") != "tool":
continue
content = msg.get("content")
if isinstance(content, str):
text = content
prose = content
else:
text = extract_text(msg)
prose = _text_only(msg)
if len(text) <= protected_max_chars:
continue
head_len = 1500
tail_len = 1500
head = prose[:head_len]
tail = prose[-tail_len:]
trimmed = len(prose) - head_len - tail_len
marker = f"\n\n[{trimmed:,} chars trimmed...]\n\n" if trimmed > 0 else ""
elided = f"{head}{marker}{tail}"
if trimmed > 0 and len(elided) < len(prose):
summarized += 1
pruned[i] = _with_text(msg, elided)
final_tokens = sum(estimate_tokens(extract_text(m)) for m in pruned)
return pruned, {
"pruned": True,

View File

@@ -3774,6 +3774,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks):
extra_fixed_tokens=_tools_overhead,
),
extra_fixed_tokens=_tools_overhead,
protected_max_chars=cfg.pinch.protected_max_chars,
)
if pinch_stats.get("pruned"):
logs.info(
@@ -4017,6 +4018,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks):
extra_fixed_tokens=_tools_overhead,
),
extra_fixed_tokens=_tools_overhead,
protected_max_chars=cfg.pinch.protected_max_chars,
)
if pinch_stats.get("pruned"):
logs.info(

View File

@@ -415,6 +415,32 @@ def test_pinch_max_summarize_chars_accepts_default(raw):
assert loaded.pinch.max_summarize_chars == 4000
def test_nonminimum_pinch_protected_max_chars_is_rejected(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["protected_max_chars"] = 100
with pytest.raises(ValueError, match="protected_max_chars.*>=.*3000"):
RouterConfig(**cfg)
def test_pinch_protected_max_chars_at_valid_min_loads(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["protected_max_chars"] = 3000
loaded = RouterConfig(**cfg)
assert loaded.pinch.protected_max_chars == 3000
def test_pinch_protected_max_chars_accepts_default(raw):
loaded = RouterConfig(**raw)
assert loaded.pinch.protected_max_chars == 20000
def test_pinch_protected_max_chars_null_disables(raw):
cfg = copy.deepcopy(raw)
cfg["pinch"]["protected_max_chars"] = None
loaded = RouterConfig(**cfg)
assert loaded.pinch.protected_max_chars is None
def test_pinch_relevance_defaults_load(raw):
loaded = RouterConfig(**raw)
assert loaded.pinch.relevance.enabled is True

View File

@@ -93,6 +93,63 @@ def test_recent_turn_is_protected_from_pruning():
assert "chars trimmed" in old_tool["content"]
def test_protected_max_chars_caps_outsized_recent_tool_result():
# A protected (recent) tool result far larger than protected_max_chars
# -- e.g. a full verbose test-suite run -- still gets elided, closing
# the gap keep_last_turns otherwise leaves: no size limit inside the
# protected window at all.
messages = [
_user("first"),
_tool("search", "old result " * 3000),
_assistant("first answer"),
_user("second"),
_tool("pytest", "PASSED " * 50000), # huge, but inside the protected window
_assistant("second answer"),
]
out, stats = prune_context(
messages, budget_tokens=100, keep_last_turns=1, protected_max_chars=2000,
)
recent_tool = [m for m in out if m.get("role") == "tool"][-1]
assert "chars trimmed" in recent_tool["content"]
assert len(recent_tool["content"]) < len(messages[4]["content"])
assert stats["tokens_saved"] > 0
def test_protected_max_chars_none_preserves_historical_behavior():
# The pure function's default (None) must reproduce the exact historical
# behavior new tests weren't written against -- a huge recent tool result
# ships verbatim when the cap isn't configured.
messages = [
_user("first"),
_tool("search", "old result " * 3000),
_assistant("first answer"),
_user("second"),
_tool("pytest", "PASSED " * 50000),
_assistant("second answer"),
]
out, _ = prune_context(messages, budget_tokens=100, keep_last_turns=1)
recent_tool = [m for m in out if m.get("role") == "tool"][-1]
assert recent_tool["content"] == messages[4]["content"]
def test_protected_max_chars_does_not_touch_small_recent_results():
# A recent tool result under the cap is left alone even when the cap is
# configured -- this only catches true outliers, never ordinary output.
messages = [
_user("first"),
_tool("search", "old result " * 3000),
_assistant("first answer"),
_user("second"),
_tool("search", "small recent result"),
_assistant("second answer"),
]
out, _ = prune_context(
messages, budget_tokens=100, keep_last_turns=1, protected_max_chars=2000,
)
recent_tool = [m for m in out if m.get("role") == "tool"][-1]
assert recent_tool["content"] == "small recent result"
def test_tool_result_role_pairing_preserved():
# The tool message keeps its name and position, so the API still parses.
messages = [