Files
6krrt/tests/test_textcut.py
adlee-was-taken fc02d770d2 fix: cut text on character boundaries, and write up the mutation-point audit
The audit prompted by 8518114: every place the router encodes, decodes or
slices content, classified by what it can actually break.
plans/text-integrity-audit.md has the table. Two findings needed code.

Truncation could cut a grapheme cluster. Five head/tail slices -- the
classifier input clamp, the local-checker elision, and three sites in
context_prune -- sliced str directly. None could produce mojibake, because
Python indexes codepoints, but all of them could split a cluster: "cafe" +
U+0301 sliced at 4 drops the accent and leaves a bare combining mark at the
head of the tail. The same goes for ZWJ emoji sequences, variation
selectors and the regional-indicator pairs that make flags.

Severity is well below the SSE bug -- a stray mark, not a mangled document
-- but the SHAPE is the one this project keeps paying for: three of those
sites produce the prompt sent to the provider, so a bad cut is the router
corrupting the model's input and then reading the model's output as though
the model were solely responsible. src/textcut.py moves the cut to the
nearest boundary instead, shrinking rather than growing so a caller's
length stays a ceiling.

Reverting textcut fails 3 of the 5 new call-site tests, which is the point
of having them separate from the unit tests: a correct helper nobody calls
prevents nothing.

The six router-generated SSE writes are safe and now say so in the audit.
They look exactly like the bug that was just fixed and differ by one
keyword -- json.dumps defaults to ensure_ascii=True, so the payload is pure
ASCII before it is encoded. Anyone passing ensure_ascii=False there to save
bytes reintroduces a charset decision on an output path.

Every fixture in these files uses \u escapes rather than literal non-ASCII.
Files about text corruption should not silently change meaning if they ever
round-trip through something that mangles encodings.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
2026-09-08 14:15:35 -04:00

120 lines
4.0 KiB
Python

"""safe_head / safe_tail must never hand back half a character.
The property under test is not "the output is short enough" -- a plain slice
already manages that. It is that neither half of a cut is a fragment that
means something different from what it was part of.
"""
from __future__ import annotations
import unicodedata
import pytest
from textcut import safe_head, safe_tail
# e + COMBINING ACUTE. Four codepoints, three characters to a reader.
ACCENT = "cafe\u0301"
# family emoji: four people joined by ZWJ
ZWJ_FAMILY = "\U0001f468\u200d\U0001f469\u200d\U0001f467\u200d\U0001f466"
# regional indicators: two flags, four codepoints
FLAGS = "\U0001f1eb\U0001f1ee\U0001f1f8\U0001f1ea"
# base + variation selector 16 (render as emoji)
VARIATION = "\u2764\ufe0f"
def _is_mark(ch: str) -> bool:
return unicodedata.category(ch) in {"Mn", "Mc", "Me"}
# --- the failure a plain slice makes ---------------------------------------
def test_a_plain_slice_orphans_a_combining_mark():
"""The bug being prevented, stated as a fact about str slicing."""
assert ACCENT[:4] == "cafe" # accent silently dropped
assert _is_mark(ACCENT[4:][0]) # tail begins with a bare accent
# --- heads ------------------------------------------------------------------
def test_head_drops_the_base_rather_than_orphaning_its_mark():
assert safe_head(ACCENT, 4) == "caf"
def test_head_keeps_a_whole_cluster_when_it_fits():
assert safe_head(ACCENT, 5) == ACCENT
def test_head_does_not_split_a_zwj_sequence():
for n in range(1, len(ZWJ_FAMILY)):
out = safe_head(ZWJ_FAMILY, n)
assert not out.endswith("\u200d"), n
assert out in ("", ZWJ_FAMILY[: len(out)])
assert safe_head(ZWJ_FAMILY, len(ZWJ_FAMILY) - 1) == ""
def test_head_does_not_split_a_flag():
# One flag is two codepoints; cutting at 3 would leave one and a half.
assert safe_head(FLAGS, 3) == FLAGS[:2]
assert safe_head(FLAGS, 2) == FLAGS[:2]
def test_head_does_not_orphan_a_variation_selector():
assert safe_head(VARIATION, 1) == ""
assert safe_head(VARIATION, 2) == VARIATION
# --- tails ------------------------------------------------------------------
def test_tail_does_not_begin_with_a_combining_mark():
out = safe_tail(ACCENT, 1)
assert out == ""
out = safe_tail(ACCENT, 2)
assert out == "e\u0301"
assert not _is_mark(out[0])
def test_tail_does_not_begin_mid_zwj_sequence():
for n in range(1, len(ZWJ_FAMILY)):
out = safe_tail(ZWJ_FAMILY, n)
assert not out.startswith("\u200d"), n
def test_tail_does_not_begin_mid_flag():
assert safe_tail(FLAGS, 3) == FLAGS[2:]
# --- invariants -------------------------------------------------------------
@pytest.mark.parametrize("text", [ACCENT, ZWJ_FAMILY, FLAGS, VARIATION,
"plain ascii", "", "a"])
def test_neither_cut_ever_exceeds_its_budget(text):
"""n is a ceiling. A cut that grew to keep a cluster whole would let
input the caller does not control blow past a budget it set."""
for n in range(0, len(text) + 3):
assert len(safe_head(text, n)) <= max(n, 0)
assert len(safe_tail(text, n)) <= max(n, 0)
@pytest.mark.parametrize("text", [ACCENT, ZWJ_FAMILY, FLAGS, VARIATION])
def test_both_cuts_are_prefixes_and_suffixes_of_the_input(text):
"""Nothing is invented or reordered -- these only ever shorten."""
for n in range(0, len(text) + 3):
assert text.startswith(safe_head(text, n))
assert text.endswith(safe_tail(text, n))
def test_ascii_is_untouched():
"""The common case must not pay for the rare one."""
s = "the quick brown fox"
for n in range(0, len(s) + 3):
assert safe_head(s, n) == s[:n] if n <= len(s) else safe_head(s, n) == s
if 0 < n <= len(s):
assert safe_tail(s, n) == s[-n:]
def test_a_full_length_request_returns_the_whole_string():
for text in (ACCENT, ZWJ_FAMILY, FLAGS, VARIATION):
assert safe_head(text, len(text)) == text
assert safe_tail(text, len(text)) == text