fix(pinch): cap outsized tool results even inside the protected window #44
@@ -282,6 +282,17 @@ pinch:
|
|||||||
budget_tokens: 50000
|
budget_tokens: 50000
|
||||||
keep_last_turns: 4
|
keep_last_turns: 4
|
||||||
max_summarize_chars: 4000
|
max_summarize_chars: 4000
|
||||||
|
# keep_last_turns has no size limit inside it: an entire autonomous
|
||||||
|
# tool-call loop with no new user message can be one protected turn, and
|
||||||
|
# one outsized tool result inside it (a full verbose test run, a huge file
|
||||||
|
# read) ships verbatim regardless of size. Measured live 2026-09-06: a
|
||||||
|
# 324k-token conversation shrank only ~8% because nearly all of it sat
|
||||||
|
# inside the protected window. This closes that gap: any tool result
|
||||||
|
# inside the protected window over this many characters still gets the
|
||||||
|
# same head/tail elision candidates get. Deliberately a much higher bar
|
||||||
|
# than max_summarize_chars -- recent results are more likely to still
|
||||||
|
# matter -- so it only catches true outliers. Set to null to disable.
|
||||||
|
protected_max_chars: 20000
|
||||||
relevance:
|
relevance:
|
||||||
# Off by default, matching every other new-and-unproven knob in this
|
# Off by default, matching every other new-and-unproven knob in this
|
||||||
# project — and specifically requires pinch.enabled too, since this has no
|
# project — and specifically requires pinch.enabled too, since this has no
|
||||||
|
|||||||
@@ -99,6 +99,7 @@ pinch:
|
|||||||
budget_tokens: 50000
|
budget_tokens: 50000
|
||||||
keep_last_turns: 4
|
keep_last_turns: 4
|
||||||
max_summarize_chars: 4000
|
max_summarize_chars: 4000
|
||||||
|
protected_max_chars: 20000
|
||||||
relevance:
|
relevance:
|
||||||
enabled: true
|
enabled: true
|
||||||
model: "nomic-embed-text"
|
model: "nomic-embed-text"
|
||||||
@@ -113,6 +114,7 @@ pinch:
|
|||||||
| `pinch.budget_tokens` | `50000` | Above this estimated token count the conversation is pruned. Must be `> 0`. |
|
| `pinch.budget_tokens` | `50000` | Above this estimated token count the conversation is pruned. Must be `> 0`. |
|
||||||
| `pinch.keep_last_turns` | `4` | How many recent user turns (plus their assistant replies and tool results) are protected from pruning. Must be `> 0`. |
|
| `pinch.keep_last_turns` | `4` | How many recent user turns (plus their assistant replies and tool results) are protected from pruning. Must be `> 0`. |
|
||||||
| `pinch.max_summarize_chars` | `4000` | Tool results longer than this many chars are summarized in place; shorter ones are collapsed to a placeholder. Must be `>= 3000` — below that summarization would grow the message. |
|
| `pinch.max_summarize_chars` | `4000` | Tool results longer than this many chars are summarized in place; shorter ones are collapsed to a placeholder. Must be `>= 3000` — below that summarization would grow the message. |
|
||||||
|
| `pinch.protected_max_chars` | `20000` | `keep_last_turns` has no size limit inside it — an entire autonomous tool-call loop with no new user message can be one protected turn, so one outsized tool result inside it (a full verbose test run, a huge file read) used to ship verbatim regardless of size. Measured live 2026-09-06: a 324k-token conversation shrank only ~8% because nearly all of it sat inside the protected window. Any tool result inside the protected window over this many chars now gets the same head/tail elision candidates get — deliberately a much higher bar than `max_summarize_chars`, since recent results are more likely to still matter, so it only catches true outliers. Must be `>= 3000`, or `null` to disable. |
|
||||||
| `pinch.relevance.enabled` | `true` | Embedding-model relevance scoring. Requires `pinch.enabled` too. Any failure reverts to uniform trimming. |
|
| `pinch.relevance.enabled` | `true` | Embedding-model relevance scoring. Requires `pinch.enabled` too. Any failure reverts to uniform trimming. |
|
||||||
| `pinch.relevance.model` | `nomic-embed-text` | An **embedding** model — never `classifier.model` or `verification.model`. |
|
| `pinch.relevance.model` | `nomic-embed-text` | An **embedding** model — never `classifier.model` or `verification.model`. |
|
||||||
| `pinch.relevance.base_url` | `http://localhost:11434/v1` | OpenAI-compatible embeddings endpoint on the same local Ollama. |
|
| `pinch.relevance.base_url` | `http://localhost:11434/v1` | OpenAI-compatible embeddings endpoint on the same local Ollama. |
|
||||||
|
|||||||
@@ -575,6 +575,15 @@ class PinchConfig(StrictModel):
|
|||||||
keep_last_turns: int = 4
|
keep_last_turns: int = 4
|
||||||
# Tool results longer than this many characters are summarized in place.
|
# Tool results longer than this many characters are summarized in place.
|
||||||
max_summarize_chars: int = 4000
|
max_summarize_chars: int = 4000
|
||||||
|
# keep_last_turns has no size limit inside it -- an entire autonomous
|
||||||
|
# tool-call loop with no new user message can be one protected turn, and
|
||||||
|
# one outsized tool result inside it (a full verbose test run, a huge
|
||||||
|
# file read) ships verbatim regardless of size. Measured live: a
|
||||||
|
# 324k-token conversation shrank only ~8% because nearly all of it sat
|
||||||
|
# inside the protected window. This is deliberately a much higher bar
|
||||||
|
# than max_summarize_chars -- recent results are more likely to still
|
||||||
|
# matter -- so it only catches true outliers. None disables it.
|
||||||
|
protected_max_chars: Optional[int] = 20000
|
||||||
relevance: PinchRelevanceConfig = PinchRelevanceConfig()
|
relevance: PinchRelevanceConfig = PinchRelevanceConfig()
|
||||||
|
|
||||||
@field_validator("budget_tokens")
|
@field_validator("budget_tokens")
|
||||||
@@ -601,6 +610,17 @@ class PinchConfig(StrictModel):
|
|||||||
)
|
)
|
||||||
return v
|
return v
|
||||||
|
|
||||||
|
@field_validator("protected_max_chars")
|
||||||
|
@classmethod
|
||||||
|
def protected_max_chars_valid(cls, v: Optional[int]) -> Optional[int]:
|
||||||
|
if v is not None and v < 3000:
|
||||||
|
raise ValueError(
|
||||||
|
"pinch.protected_max_chars must be >= 3000, or null to "
|
||||||
|
"disable (below this, elision grows the message, same "
|
||||||
|
"floor as max_summarize_chars)"
|
||||||
|
)
|
||||||
|
return v
|
||||||
|
|
||||||
|
|
||||||
class SessionCacheConfig(StrictModel):
|
class SessionCacheConfig(StrictModel):
|
||||||
"""Per-session classification cache (in-memory, process lifetime).
|
"""Per-session classification cache (in-memory, process lifetime).
|
||||||
|
|||||||
@@ -243,6 +243,7 @@ def prune_context(
|
|||||||
max_summarize_chars: int = PinchConfig.model_fields["max_summarize_chars"].default,
|
max_summarize_chars: int = PinchConfig.model_fields["max_summarize_chars"].default,
|
||||||
relevance_order: Optional[list[int]] = None,
|
relevance_order: Optional[list[int]] = None,
|
||||||
extra_fixed_tokens: int = 0,
|
extra_fixed_tokens: int = 0,
|
||||||
|
protected_max_chars: Optional[int] = None,
|
||||||
) -> tuple[list[dict], dict]:
|
) -> tuple[list[dict], dict]:
|
||||||
"""Trim old tool results once a conversation exceeds ``budget_tokens``.
|
"""Trim old tool results once a conversation exceeds ``budget_tokens``.
|
||||||
|
|
||||||
@@ -260,6 +261,22 @@ def prune_context(
|
|||||||
|
|
||||||
Only runs (and only mutates anything) when the estimate actually exceeds
|
Only runs (and only mutates anything) when the estimate actually exceeds
|
||||||
the budget; otherwise the original list is returned untouched.
|
the budget; otherwise the original list is returned untouched.
|
||||||
|
|
||||||
|
``protected_max_chars`` closes a real gap in the ``keep_last_turns``
|
||||||
|
guarantee: that window protects everything from the last N *user-turn*
|
||||||
|
boundaries onward, with no size limit inside it. In a long autonomous
|
||||||
|
tool-call loop (no new user message between calls), an entire multi-hour
|
||||||
|
session can be a single protected turn, and one outsized tool result
|
||||||
|
inside it — a full verbose test-suite run, a huge file read — ships
|
||||||
|
verbatim no matter how large. Measured live: a 324k-token conversation
|
||||||
|
shrank only ~8% because nearly all of it sat inside the protected window.
|
||||||
|
When set, any tool result inside the protected window whose text exceeds
|
||||||
|
this many characters still gets the same head/tail elision candidates get
|
||||||
|
-- deliberately a much higher bar than ``max_summarize_chars`` (recent
|
||||||
|
results are more likely to matter), so this only catches true outliers,
|
||||||
|
never ordinary recent tool output. ``None`` (the pure function's default)
|
||||||
|
preserves the exact historical behavior; ``PinchConfig`` supplies a real
|
||||||
|
default so production gets the fix without every caller needing to opt in.
|
||||||
"""
|
"""
|
||||||
orig_tokens = sum(estimate_tokens(extract_text(m)) for m in messages) + extra_fixed_tokens
|
orig_tokens = sum(estimate_tokens(extract_text(m)) for m in messages) + extra_fixed_tokens
|
||||||
if orig_tokens <= budget_tokens:
|
if orig_tokens <= budget_tokens:
|
||||||
@@ -387,6 +404,31 @@ def prune_context(
|
|||||||
else:
|
else:
|
||||||
pruned.append(msg)
|
pruned.append(msg)
|
||||||
|
|
||||||
|
if protected_max_chars is not None:
|
||||||
|
for i in range(protected_from, len(pruned)):
|
||||||
|
msg = pruned[i]
|
||||||
|
if msg.get("role") != "tool":
|
||||||
|
continue
|
||||||
|
content = msg.get("content")
|
||||||
|
if isinstance(content, str):
|
||||||
|
text = content
|
||||||
|
prose = content
|
||||||
|
else:
|
||||||
|
text = extract_text(msg)
|
||||||
|
prose = _text_only(msg)
|
||||||
|
if len(text) <= protected_max_chars:
|
||||||
|
continue
|
||||||
|
head_len = 1500
|
||||||
|
tail_len = 1500
|
||||||
|
head = prose[:head_len]
|
||||||
|
tail = prose[-tail_len:]
|
||||||
|
trimmed = len(prose) - head_len - tail_len
|
||||||
|
marker = f"\n\n[{trimmed:,} chars trimmed...]\n\n" if trimmed > 0 else ""
|
||||||
|
elided = f"{head}{marker}{tail}"
|
||||||
|
if trimmed > 0 and len(elided) < len(prose):
|
||||||
|
summarized += 1
|
||||||
|
pruned[i] = _with_text(msg, elided)
|
||||||
|
|
||||||
final_tokens = sum(estimate_tokens(extract_text(m)) for m in pruned)
|
final_tokens = sum(estimate_tokens(extract_text(m)) for m in pruned)
|
||||||
return pruned, {
|
return pruned, {
|
||||||
"pruned": True,
|
"pruned": True,
|
||||||
|
|||||||
@@ -3774,6 +3774,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks):
|
|||||||
extra_fixed_tokens=_tools_overhead,
|
extra_fixed_tokens=_tools_overhead,
|
||||||
),
|
),
|
||||||
extra_fixed_tokens=_tools_overhead,
|
extra_fixed_tokens=_tools_overhead,
|
||||||
|
protected_max_chars=cfg.pinch.protected_max_chars,
|
||||||
)
|
)
|
||||||
if pinch_stats.get("pruned"):
|
if pinch_stats.get("pruned"):
|
||||||
logs.info(
|
logs.info(
|
||||||
@@ -4017,6 +4018,7 @@ def chat_completions(body: dict[str, Any], background: BackgroundTasks):
|
|||||||
extra_fixed_tokens=_tools_overhead,
|
extra_fixed_tokens=_tools_overhead,
|
||||||
),
|
),
|
||||||
extra_fixed_tokens=_tools_overhead,
|
extra_fixed_tokens=_tools_overhead,
|
||||||
|
protected_max_chars=cfg.pinch.protected_max_chars,
|
||||||
)
|
)
|
||||||
if pinch_stats.get("pruned"):
|
if pinch_stats.get("pruned"):
|
||||||
logs.info(
|
logs.info(
|
||||||
|
|||||||
@@ -415,6 +415,32 @@ def test_pinch_max_summarize_chars_accepts_default(raw):
|
|||||||
assert loaded.pinch.max_summarize_chars == 4000
|
assert loaded.pinch.max_summarize_chars == 4000
|
||||||
|
|
||||||
|
|
||||||
|
def test_nonminimum_pinch_protected_max_chars_is_rejected(raw):
|
||||||
|
cfg = copy.deepcopy(raw)
|
||||||
|
cfg["pinch"]["protected_max_chars"] = 100
|
||||||
|
with pytest.raises(ValueError, match="protected_max_chars.*>=.*3000"):
|
||||||
|
RouterConfig(**cfg)
|
||||||
|
|
||||||
|
|
||||||
|
def test_pinch_protected_max_chars_at_valid_min_loads(raw):
|
||||||
|
cfg = copy.deepcopy(raw)
|
||||||
|
cfg["pinch"]["protected_max_chars"] = 3000
|
||||||
|
loaded = RouterConfig(**cfg)
|
||||||
|
assert loaded.pinch.protected_max_chars == 3000
|
||||||
|
|
||||||
|
|
||||||
|
def test_pinch_protected_max_chars_accepts_default(raw):
|
||||||
|
loaded = RouterConfig(**raw)
|
||||||
|
assert loaded.pinch.protected_max_chars == 20000
|
||||||
|
|
||||||
|
|
||||||
|
def test_pinch_protected_max_chars_null_disables(raw):
|
||||||
|
cfg = copy.deepcopy(raw)
|
||||||
|
cfg["pinch"]["protected_max_chars"] = None
|
||||||
|
loaded = RouterConfig(**cfg)
|
||||||
|
assert loaded.pinch.protected_max_chars is None
|
||||||
|
|
||||||
|
|
||||||
def test_pinch_relevance_defaults_load(raw):
|
def test_pinch_relevance_defaults_load(raw):
|
||||||
loaded = RouterConfig(**raw)
|
loaded = RouterConfig(**raw)
|
||||||
assert loaded.pinch.relevance.enabled is True
|
assert loaded.pinch.relevance.enabled is True
|
||||||
|
|||||||
@@ -93,6 +93,63 @@ def test_recent_turn_is_protected_from_pruning():
|
|||||||
assert "chars trimmed" in old_tool["content"]
|
assert "chars trimmed" in old_tool["content"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_protected_max_chars_caps_outsized_recent_tool_result():
|
||||||
|
# A protected (recent) tool result far larger than protected_max_chars
|
||||||
|
# -- e.g. a full verbose test-suite run -- still gets elided, closing
|
||||||
|
# the gap keep_last_turns otherwise leaves: no size limit inside the
|
||||||
|
# protected window at all.
|
||||||
|
messages = [
|
||||||
|
_user("first"),
|
||||||
|
_tool("search", "old result " * 3000),
|
||||||
|
_assistant("first answer"),
|
||||||
|
_user("second"),
|
||||||
|
_tool("pytest", "PASSED " * 50000), # huge, but inside the protected window
|
||||||
|
_assistant("second answer"),
|
||||||
|
]
|
||||||
|
out, stats = prune_context(
|
||||||
|
messages, budget_tokens=100, keep_last_turns=1, protected_max_chars=2000,
|
||||||
|
)
|
||||||
|
recent_tool = [m for m in out if m.get("role") == "tool"][-1]
|
||||||
|
assert "chars trimmed" in recent_tool["content"]
|
||||||
|
assert len(recent_tool["content"]) < len(messages[4]["content"])
|
||||||
|
assert stats["tokens_saved"] > 0
|
||||||
|
|
||||||
|
|
||||||
|
def test_protected_max_chars_none_preserves_historical_behavior():
|
||||||
|
# The pure function's default (None) must reproduce the exact historical
|
||||||
|
# behavior new tests weren't written against -- a huge recent tool result
|
||||||
|
# ships verbatim when the cap isn't configured.
|
||||||
|
messages = [
|
||||||
|
_user("first"),
|
||||||
|
_tool("search", "old result " * 3000),
|
||||||
|
_assistant("first answer"),
|
||||||
|
_user("second"),
|
||||||
|
_tool("pytest", "PASSED " * 50000),
|
||||||
|
_assistant("second answer"),
|
||||||
|
]
|
||||||
|
out, _ = prune_context(messages, budget_tokens=100, keep_last_turns=1)
|
||||||
|
recent_tool = [m for m in out if m.get("role") == "tool"][-1]
|
||||||
|
assert recent_tool["content"] == messages[4]["content"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_protected_max_chars_does_not_touch_small_recent_results():
|
||||||
|
# A recent tool result under the cap is left alone even when the cap is
|
||||||
|
# configured -- this only catches true outliers, never ordinary output.
|
||||||
|
messages = [
|
||||||
|
_user("first"),
|
||||||
|
_tool("search", "old result " * 3000),
|
||||||
|
_assistant("first answer"),
|
||||||
|
_user("second"),
|
||||||
|
_tool("search", "small recent result"),
|
||||||
|
_assistant("second answer"),
|
||||||
|
]
|
||||||
|
out, _ = prune_context(
|
||||||
|
messages, budget_tokens=100, keep_last_turns=1, protected_max_chars=2000,
|
||||||
|
)
|
||||||
|
recent_tool = [m for m in out if m.get("role") == "tool"][-1]
|
||||||
|
assert recent_tool["content"] == "small recent result"
|
||||||
|
|
||||||
|
|
||||||
def test_tool_result_role_pairing_preserved():
|
def test_tool_result_role_pairing_preserved():
|
||||||
# The tool message keeps its name and position, so the API still parses.
|
# The tool message keeps its name and position, so the API still parses.
|
||||||
messages = [
|
messages = [
|
||||||
|
|||||||
Reference in New Issue
Block a user