"""Every place the router shortens text must cut on a character boundary. `textcut` is unit-tested next door; this file tests the CALL SITES, because a correct helper that nobody calls prevents nothing. Each test drives the real function with input engineered so a naive `[:n]` / `[-n:]` would land inside a grapheme cluster, and asserts the router did not hand on a fragment. The decomposed form matters. "e\\u0301" is two codepoints that render as one character; the precomposed "\\u00e9" is one codepoint and nothing can split it. A test written with the precomposed form passes against a broken implementation, which is the same way an all-ASCII fixture let a charset bug survive a test named "proxied verbatim". """ from __future__ import annotations import unicodedata import dispatcher import verification from context_prune import prune_context E_ACUTE = "e\u0301" # decomposed: LATIN SMALL LETTER E + COMBINING ACUTE def _is_mark(ch: str) -> bool: return unicodedata.category(ch) in {"Mn", "Mc", "Me"} def _assert_no_orphan_mark(text: str, where: str) -> None: assert text, f"{where}: produced nothing to check" assert not _is_mark(text[0]), ( f"{where}: fragment begins with a bare combining mark " f"{text[0]!r} -- a cluster was cut in half" ) def test_the_decomposed_form_is_what_a_naive_slice_breaks(): """The premise, asserted rather than assumed.""" s = "caf" + E_ACUTE assert len(s) == 5 assert s[:4] == "cafe" # accent silently dropped assert _is_mark(s[4:][0]) # remainder is a bare accent # ...and the precomposed form cannot reproduce it, which is why this file # never uses it. assert len("caf\u00e9") == 4 def test_clamp_for_classifier_does_not_orphan_a_mark(): """The classifier input clamp cuts head AND tail.""" max_chars = 40 half = max_chars // 2 # Put the combining mark exactly where the tail cut would land. text = "a" * 500 + E_ACUTE + "b" * (half - 1) out = dispatcher.clamp_for_classifier(text, max_chars) tail = out.rsplit("...]\n\n", 1)[-1] _assert_no_orphan_mark(tail, "clamp_for_classifier tail") assert len(out) < len(text) def test_excerpt_does_not_orphan_a_mark(): """The local checker judges the ending, so the ending must be whole.""" limit = 400 tail_len = limit - limit // 2 text = "x" * 2000 + E_ACUTE + "y" * (tail_len - 1) out = verification.excerpt(text, limit) tail = out.rsplit("...]\n\n", 1)[-1] _assert_no_orphan_mark(tail, "excerpt tail") def test_prune_context_does_not_orphan_a_mark_in_the_prompt(): """This one is the prompt itself -- a bad cut corrupts the model's input. That is the case that matters most: the router mangles what it sends, the model faithfully reproduces the mangling, and the output looks like the model's fault. """ # prune_context elides with head_len = tail_len = 1500, so place the # cluster so the tail cut falls between its two codepoints. big = "z" * 40000 + E_ACUTE + "w" * 1499 # Six trailing messages, because keep_last_turns defaults to 4 and a tool # result inside that window is kept verbatim -- the elision path never # runs and the test would pass without testing anything. messages = ( [{"role": "user", "content": "first"}, {"role": "tool", "content": big}] + [{"role": "user", "content": f"m{i}"} for i in range(6)] ) pruned, _stats = prune_context(messages, budget_tokens=200) elided = [ m["content"] for m in pruned if isinstance(m.get("content"), str) and "chars trimmed" in m["content"] ] assert elided, "fixture did not trigger the head/tail elision path" for content in elided: tail = content.rsplit("chars trimmed...]\n\n", 1)[-1] _assert_no_orphan_mark(tail, "prune_context tail") def test_prune_context_reports_a_truthful_trimmed_count(): """The marker counts what was actually dropped. A cluster-safe cut removes one or two characters more than the nominal budget, so a count computed from head_len/tail_len rather than from the real head and tail would be quietly wrong. """ big = "z" * 40000 + E_ACUTE + "w" * 1499 # Six trailing messages, because keep_last_turns defaults to 4 and a tool # result inside that window is kept verbatim -- the elision path never # runs and the test would pass without testing anything. messages = ( [{"role": "user", "content": "first"}, {"role": "tool", "content": big}] + [{"role": "user", "content": f"m{i}"} for i in range(6)] ) pruned, _stats = prune_context(messages, budget_tokens=200) for m in pruned: content = m.get("content") if not isinstance(content, str) or "chars trimmed" not in content: continue head, rest = content.split("\n\n[", 1) stated = int(rest.split(" chars trimmed", 1)[0].replace(",", "")) tail = rest.split("chars trimmed...]\n\n", 1)[-1] assert stated == len(big) - len(head) - len(tail) return raise AssertionError("fixture did not trigger the head/tail elision path")