The audit prompted by 8518114: every place the router encodes, decodes or
slices content, classified by what it can actually break.
plans/text-integrity-audit.md has the table. Two findings needed code.
Truncation could cut a grapheme cluster. Five head/tail slices -- the
classifier input clamp, the local-checker elision, and three sites in
context_prune -- sliced str directly. None could produce mojibake, because
Python indexes codepoints, but all of them could split a cluster: "cafe" +
U+0301 sliced at 4 drops the accent and leaves a bare combining mark at the
head of the tail. The same goes for ZWJ emoji sequences, variation
selectors and the regional-indicator pairs that make flags.
Severity is well below the SSE bug -- a stray mark, not a mangled document
-- but the SHAPE is the one this project keeps paying for: three of those
sites produce the prompt sent to the provider, so a bad cut is the router
corrupting the model's input and then reading the model's output as though
the model were solely responsible. src/textcut.py moves the cut to the
nearest boundary instead, shrinking rather than growing so a caller's
length stays a ceiling.
Reverting textcut fails 3 of the 5 new call-site tests, which is the point
of having them separate from the unit tests: a correct helper nobody calls
prevents nothing.
The six router-generated SSE writes are safe and now say so in the audit.
They look exactly like the bug that was just fixed and differ by one
keyword -- json.dumps defaults to ensure_ascii=True, so the payload is pure
ASCII before it is encoded. Anyone passing ensure_ascii=False there to save
bytes reintroduces a charset decision on an output path.
Every fixture in these files uses \u escapes rather than literal non-ASCII.
Files about text corruption should not silently change meaning if they ever
round-trip through something that mangles encodings.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VRQXz5SYZYVWscxS1QqF6U
120 lines
4.0 KiB
Python
120 lines
4.0 KiB
Python
"""safe_head / safe_tail must never hand back half a character.
|
|
|
|
The property under test is not "the output is short enough" -- a plain slice
|
|
already manages that. It is that neither half of a cut is a fragment that
|
|
means something different from what it was part of.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import unicodedata
|
|
|
|
import pytest
|
|
|
|
from textcut import safe_head, safe_tail
|
|
|
|
# e + COMBINING ACUTE. Four codepoints, three characters to a reader.
|
|
ACCENT = "cafe\u0301"
|
|
# family emoji: four people joined by ZWJ
|
|
ZWJ_FAMILY = "\U0001f468\u200d\U0001f469\u200d\U0001f467\u200d\U0001f466"
|
|
# regional indicators: two flags, four codepoints
|
|
FLAGS = "\U0001f1eb\U0001f1ee\U0001f1f8\U0001f1ea"
|
|
# base + variation selector 16 (render as emoji)
|
|
VARIATION = "\u2764\ufe0f"
|
|
|
|
|
|
def _is_mark(ch: str) -> bool:
|
|
return unicodedata.category(ch) in {"Mn", "Mc", "Me"}
|
|
|
|
|
|
# --- the failure a plain slice makes ---------------------------------------
|
|
|
|
def test_a_plain_slice_orphans_a_combining_mark():
|
|
"""The bug being prevented, stated as a fact about str slicing."""
|
|
assert ACCENT[:4] == "cafe" # accent silently dropped
|
|
assert _is_mark(ACCENT[4:][0]) # tail begins with a bare accent
|
|
|
|
|
|
# --- heads ------------------------------------------------------------------
|
|
|
|
def test_head_drops_the_base_rather_than_orphaning_its_mark():
|
|
assert safe_head(ACCENT, 4) == "caf"
|
|
|
|
|
|
def test_head_keeps_a_whole_cluster_when_it_fits():
|
|
assert safe_head(ACCENT, 5) == ACCENT
|
|
|
|
|
|
def test_head_does_not_split_a_zwj_sequence():
|
|
for n in range(1, len(ZWJ_FAMILY)):
|
|
out = safe_head(ZWJ_FAMILY, n)
|
|
assert not out.endswith("\u200d"), n
|
|
assert out in ("", ZWJ_FAMILY[: len(out)])
|
|
assert safe_head(ZWJ_FAMILY, len(ZWJ_FAMILY) - 1) == ""
|
|
|
|
|
|
def test_head_does_not_split_a_flag():
|
|
# One flag is two codepoints; cutting at 3 would leave one and a half.
|
|
assert safe_head(FLAGS, 3) == FLAGS[:2]
|
|
assert safe_head(FLAGS, 2) == FLAGS[:2]
|
|
|
|
|
|
def test_head_does_not_orphan_a_variation_selector():
|
|
assert safe_head(VARIATION, 1) == ""
|
|
assert safe_head(VARIATION, 2) == VARIATION
|
|
|
|
|
|
# --- tails ------------------------------------------------------------------
|
|
|
|
def test_tail_does_not_begin_with_a_combining_mark():
|
|
out = safe_tail(ACCENT, 1)
|
|
assert out == ""
|
|
out = safe_tail(ACCENT, 2)
|
|
assert out == "e\u0301"
|
|
assert not _is_mark(out[0])
|
|
|
|
|
|
def test_tail_does_not_begin_mid_zwj_sequence():
|
|
for n in range(1, len(ZWJ_FAMILY)):
|
|
out = safe_tail(ZWJ_FAMILY, n)
|
|
assert not out.startswith("\u200d"), n
|
|
|
|
|
|
def test_tail_does_not_begin_mid_flag():
|
|
assert safe_tail(FLAGS, 3) == FLAGS[2:]
|
|
|
|
|
|
# --- invariants -------------------------------------------------------------
|
|
|
|
@pytest.mark.parametrize("text", [ACCENT, ZWJ_FAMILY, FLAGS, VARIATION,
|
|
"plain ascii", "", "a"])
|
|
def test_neither_cut_ever_exceeds_its_budget(text):
|
|
"""n is a ceiling. A cut that grew to keep a cluster whole would let
|
|
input the caller does not control blow past a budget it set."""
|
|
for n in range(0, len(text) + 3):
|
|
assert len(safe_head(text, n)) <= max(n, 0)
|
|
assert len(safe_tail(text, n)) <= max(n, 0)
|
|
|
|
|
|
@pytest.mark.parametrize("text", [ACCENT, ZWJ_FAMILY, FLAGS, VARIATION])
|
|
def test_both_cuts_are_prefixes_and_suffixes_of_the_input(text):
|
|
"""Nothing is invented or reordered -- these only ever shorten."""
|
|
for n in range(0, len(text) + 3):
|
|
assert text.startswith(safe_head(text, n))
|
|
assert text.endswith(safe_tail(text, n))
|
|
|
|
|
|
def test_ascii_is_untouched():
|
|
"""The common case must not pay for the rare one."""
|
|
s = "the quick brown fox"
|
|
for n in range(0, len(s) + 3):
|
|
assert safe_head(s, n) == s[:n] if n <= len(s) else safe_head(s, n) == s
|
|
if 0 < n <= len(s):
|
|
assert safe_tail(s, n) == s[-n:]
|
|
|
|
|
|
def test_a_full_length_request_returns_the_whole_string():
|
|
for text in (ACCENT, ZWJ_FAMILY, FLAGS, VARIATION):
|
|
assert safe_head(text, len(text)) == text
|
|
assert safe_tail(text, len(text)) == text
|