fix(pinch): cap outsized tool results even inside the protected window #44
@@ -282,6 +282,17 @@ pinch:
|
||||
budget_tokens: 50000
|
||||
keep_last_turns: 4
|
||||
max_summarize_chars: 4000
|
||||
# keep_last_turns has no size limit inside it: an entire autonomous
|
||||
# tool-call loop with no new user message can be one protected turn, and
|
||||
# one outsized tool result inside it (a full verbose test run, a huge file
|
||||
# read) ships verbatim regardless of size. Measured live 2026-09-06: a
|
||||
# 324k-token conversation shrank only ~8% because nearly all of it sat
|
||||
# inside the protected window. This closes that gap: any tool result
|
||||
# inside the protected window over this many characters still gets the
|
||||
# same head/tail elision candidates get. Deliberately a much higher bar
|
||||
# than max_summarize_chars -- recent results are more likely to still
|
||||
# matter -- so it only catches true outliers. Set to null to disable.
|
||||
protected_max_chars: 20000
|
||||
relevance:
|
||||
# Off by default, matching every other new-and-unproven knob in this
|
||||
# project — and specifically requires pinch.enabled too, since this has no
|
||||
|
||||
@@ -99,6 +99,7 @@ pinch:
|
||||
budget_tokens: 50000
|
||||
keep_last_turns: 4
|
||||
max_summarize_chars: 4000
|
||||
protected_max_chars: 20000
|
||||
relevance:
|
||||
enabled: true
|
||||
model: "nomic-embed-text"
|
||||
@@ -113,6 +114,7 @@ pinch:
|
||||
| `pinch.budget_tokens` | `50000` | Above this estimated token count the conversation is pruned. Must be `> 0`. |
|
||||
| `pinch.keep_last_turns` | `4` | How many recent user turns (plus their assistant replies and tool results) are protected from pruning. Must be `> 0`. |
|
||||
| `pinch.max_summarize_chars` | `4000` | Tool results longer than this many chars are summarized in place; shorter ones are collapsed to a placeholder. Must be `>= 3000` — below that summarization would grow the message. |
|
||||
| `pinch.protected_max_chars` | `20000` | `keep_last_turns` has no size limit inside it — an entire autonomous tool-call loop with no new user message can be one protected turn, so one outsized tool result inside it (a full verbose test run, a huge file read) used to ship verbatim regardless of size. Measured live 2026-09-06: a 324k-token conversation shrank only ~8% because nearly all of it sat inside the protected window. Any tool result inside the protected window over this many chars now gets the same head/tail elision candidates get — deliberately a much higher bar than `max_summarize_chars`, since recent results are more likely to still matter, so it only catches true outliers. Must be `>= 3000`, or `null` to disable. |
|
||||
| `pinch.relevance.enabled` | `true` | Embedding-model relevance scoring. Requires `pinch.enabled` too. Any failure reverts to uniform trimming. |
|
||||
| `pinch.relevance.model` | `nomic-embed-text` | An **embedding** model — never `classifier.model` or `verification.model`. |
|
||||
| `pinch.relevance.base_url` | `http://localhost:11434/v1` | OpenAI-compatible embeddings endpoint on the same local Ollama. |
|
||||
|
||||
@@ -575,6 +575,15 @@ class PinchConfig(StrictModel):
|
||||
keep_last_turns: int = 4
|
||||
# Tool results longer than this many characters are summarized in place.
|
||||
max_summarize_chars: int = 4000
|
||||
# keep_last_turns has no size limit inside it -- an entire autonomous
|
||||
# tool-call loop with no new user message can be one protected turn, and
|
||||
# one outsized tool result inside it (a full verbose test run, a huge
|
||||
# file read) ships verbatim regardless of size. Measured live: a
|
||||
# 324k-token conversation shrank only ~8% because nearly all of it sat
|
||||
# inside the protected window. This is deliberately a much higher bar
|
||||
# than max_summarize_chars -- recent results are more likely to still
|
||||
# matter -- so it only catches true outliers. None disables it.
|
||||
protected_max_chars: Optional[int] = 20000
|
||||
relevance: PinchRelevanceConfig = PinchRelevanceConfig()
|
||||
|
||||
@field_validator("budget_tokens")
|
||||
@@ -601,6 +610,17 @@ class PinchConfig(StrictModel):
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator("protected_max_chars")
|
||||
@classmethod
|
||||
def protected_max_chars_valid(cls, v: Optional[int]) -> Optional[int]:
|
||||
if v is not None and v < 3000:
|
||||
raise ValueError(
|
||||
"pinch.protected_max_chars must be >= 3000, or null to "
|
||||
"disable (below this, elision grows the message, same "
|
||||
"floor as max_summarize_chars)"
|
||||
)
|
||||
return v
|
||||
|
||||
|
||||
class SessionCacheConfig(StrictModel):
|
||||
"""Per-session classification cache (in-memory, process lifetime).
|
||||
|
||||
@@ -243,6 +243,7 @@ def prune_context(
|
||||
max_summarize_chars: int = PinchConfig.model_fields["max_summarize_chars"].default,
|
||||
relevance_order: Optional[list[int]] = None,
|
||||
extra_fixed_tokens: int = 0,
|
||||
protected_max_chars: Optional[int] = None,
|
||||
) -> tuple[list[dict], dict]:
|
||||
"""Trim old tool results once a conversation exceeds ``budget_tokens``.
|
||||
|
||||
@@ -260,6 +261,22 @@ def prune_context(
|
||||
|
||||
Only runs (and only mutates anything) when the estimate actually exceeds
|
||||
the budget; otherwise the original list is returned untouched.
|
||||
|
||||
``protected_max_chars`` closes a real gap in the ``keep_last_turns``
|
||||
guarantee: that window protects everything from the last N *user-turn*
|
||||
boundaries onward, with no size limit inside it. In a long autonomous
|
||||
tool-call loop (no new user message between calls), an entire multi-hour
|
||||
session can be a single protected turn, and one outsized tool result
|
||||
inside it — a full verbose test-suite run, a huge file read — ships
|
||||
verbatim no matter how large. Measured live: a 324k-token conversation
|
||||
shrank only ~8% because nearly all of it sat inside the protected window.
|
||||
When set, any tool result inside the protected window whose text exceeds
|
||||
this many characters still gets the same head/tail elision candidates get
|
||||
-- deliberately a much higher bar than ``max_summarize_chars`` (recent
|
||||
results are more likely to matter), so this only catches true outliers,
|
||||
never ordinary recent tool output. ``None`` (the pure function's default)
|
||||
preserves the exact historical behavior; ``PinchConfig`` supplies a real
|
||||
default so production gets the fix without every caller needing to opt in.
|
||||
"""
|
||||
orig_tokens = sum(estimate_tokens(extract_text(m)) for m in messages) + extra_fixed_tokens
|
||||
if orig_tokens <= budget_tokens:
|
||||
@@ -387,6 +404,31 @@ def prune_context(
|
||||
else:
|
||||
pruned.append(msg)
|
||||
|
||||
if protected_max_chars is not None:
|
||||
for i in range(protected_from, len(pruned)):
|
||||
msg = pruned[i]
|
||||
if msg.get("role") != "tool":
|
||||
continue
|
||||
content = msg.get("content")
|
||||
if isinstance(content, str):
|
||||
text = content
|
||||
prose = content
|
||||
else:
|
||||
text = extract_text(msg)
|
||||
prose = _text_only(msg)
|
||||
if len(text) <= protected_max_chars:
|
||||
continue
|
||||
head_len = 1500
|
||||
tail_len = 1500
|
||||
head = prose[:head_len]
|
||||
tail = prose[-tail_len:]
|
||||
trimmed = len(prose) - head_len - tail_len
|
||||
marker = f"\n\n[{trimmed:,} chars trimmed...]\n\n" if trimmed > 0 else ""
|
||||
elided = f"{head}{marker}{tail}"
|
||||
if trimmed > 0 and len(elided) < len(prose):
|
||||
summarized += 1
|
||||
pruned[i] = _with_text(msg, elided)
|
||||
|
||||
final_tokens = sum(estimate_tokens(extract_text(m)) for m in pruned)
|
||||
return pruned, {
|
||||
"pruned": True,
|
||||
|
||||
@@ -3774,6 +3774,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks):
|
||||
extra_fixed_tokens=_tools_overhead,
|
||||
),
|
||||
extra_fixed_tokens=_tools_overhead,
|
||||
protected_max_chars=cfg.pinch.protected_max_chars,
|
||||
)
|
||||
if pinch_stats.get("pruned"):
|
||||
logs.info(
|
||||
@@ -4017,6 +4018,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks):
|
||||
extra_fixed_tokens=_tools_overhead,
|
||||
),
|
||||
extra_fixed_tokens=_tools_overhead,
|
||||
protected_max_chars=cfg.pinch.protected_max_chars,
|
||||
)
|
||||
if pinch_stats.get("pruned"):
|
||||
logs.info(
|
||||
|
||||
@@ -415,6 +415,32 @@ def test_pinch_max_summarize_chars_accepts_default(raw):
|
||||
assert loaded.pinch.max_summarize_chars == 4000
|
||||
|
||||
|
||||
def test_nonminimum_pinch_protected_max_chars_is_rejected(raw):
|
||||
cfg = copy.deepcopy(raw)
|
||||
cfg["pinch"]["protected_max_chars"] = 100
|
||||
with pytest.raises(ValueError, match="protected_max_chars.*>=.*3000"):
|
||||
RouterConfig(**cfg)
|
||||
|
||||
|
||||
def test_pinch_protected_max_chars_at_valid_min_loads(raw):
|
||||
cfg = copy.deepcopy(raw)
|
||||
cfg["pinch"]["protected_max_chars"] = 3000
|
||||
loaded = RouterConfig(**cfg)
|
||||
assert loaded.pinch.protected_max_chars == 3000
|
||||
|
||||
|
||||
def test_pinch_protected_max_chars_accepts_default(raw):
|
||||
loaded = RouterConfig(**raw)
|
||||
assert loaded.pinch.protected_max_chars == 20000
|
||||
|
||||
|
||||
def test_pinch_protected_max_chars_null_disables(raw):
|
||||
cfg = copy.deepcopy(raw)
|
||||
cfg["pinch"]["protected_max_chars"] = None
|
||||
loaded = RouterConfig(**cfg)
|
||||
assert loaded.pinch.protected_max_chars is None
|
||||
|
||||
|
||||
def test_pinch_relevance_defaults_load(raw):
|
||||
loaded = RouterConfig(**raw)
|
||||
assert loaded.pinch.relevance.enabled is True
|
||||
|
||||
@@ -93,6 +93,63 @@ def test_recent_turn_is_protected_from_pruning():
|
||||
assert "chars trimmed" in old_tool["content"]
|
||||
|
||||
|
||||
def test_protected_max_chars_caps_outsized_recent_tool_result():
|
||||
# A protected (recent) tool result far larger than protected_max_chars
|
||||
# -- e.g. a full verbose test-suite run -- still gets elided, closing
|
||||
# the gap keep_last_turns otherwise leaves: no size limit inside the
|
||||
# protected window at all.
|
||||
messages = [
|
||||
_user("first"),
|
||||
_tool("search", "old result " * 3000),
|
||||
_assistant("first answer"),
|
||||
_user("second"),
|
||||
_tool("pytest", "PASSED " * 50000), # huge, but inside the protected window
|
||||
_assistant("second answer"),
|
||||
]
|
||||
out, stats = prune_context(
|
||||
messages, budget_tokens=100, keep_last_turns=1, protected_max_chars=2000,
|
||||
)
|
||||
recent_tool = [m for m in out if m.get("role") == "tool"][-1]
|
||||
assert "chars trimmed" in recent_tool["content"]
|
||||
assert len(recent_tool["content"]) < len(messages[4]["content"])
|
||||
assert stats["tokens_saved"] > 0
|
||||
|
||||
|
||||
def test_protected_max_chars_none_preserves_historical_behavior():
|
||||
# The pure function's default (None) must reproduce the exact historical
|
||||
# behavior new tests weren't written against -- a huge recent tool result
|
||||
# ships verbatim when the cap isn't configured.
|
||||
messages = [
|
||||
_user("first"),
|
||||
_tool("search", "old result " * 3000),
|
||||
_assistant("first answer"),
|
||||
_user("second"),
|
||||
_tool("pytest", "PASSED " * 50000),
|
||||
_assistant("second answer"),
|
||||
]
|
||||
out, _ = prune_context(messages, budget_tokens=100, keep_last_turns=1)
|
||||
recent_tool = [m for m in out if m.get("role") == "tool"][-1]
|
||||
assert recent_tool["content"] == messages[4]["content"]
|
||||
|
||||
|
||||
def test_protected_max_chars_does_not_touch_small_recent_results():
|
||||
# A recent tool result under the cap is left alone even when the cap is
|
||||
# configured -- this only catches true outliers, never ordinary output.
|
||||
messages = [
|
||||
_user("first"),
|
||||
_tool("search", "old result " * 3000),
|
||||
_assistant("first answer"),
|
||||
_user("second"),
|
||||
_tool("search", "small recent result"),
|
||||
_assistant("second answer"),
|
||||
]
|
||||
out, _ = prune_context(
|
||||
messages, budget_tokens=100, keep_last_turns=1, protected_max_chars=2000,
|
||||
)
|
||||
recent_tool = [m for m in out if m.get("role") == "tool"][-1]
|
||||
assert recent_tool["content"] == "small recent result"
|
||||
|
||||
|
||||
def test_tool_result_role_pairing_preserved():
|
||||
# The tool message keeps its name and position, so the API still parses.
|
||||
messages = [
|
||||
|
||||
Reference in New Issue
Block a user