"""safe_head / safe_tail must never hand back half a character. The property under test is not "the output is short enough" -- a plain slice already manages that. It is that neither half of a cut is a fragment that means something different from what it was part of. """ from __future__ import annotations import unicodedata import pytest from textcut import safe_head, safe_tail # e + COMBINING ACUTE. Four codepoints, three characters to a reader. ACCENT = "cafe\u0301" # family emoji: four people joined by ZWJ ZWJ_FAMILY = "\U0001f468\u200d\U0001f469\u200d\U0001f467\u200d\U0001f466" # regional indicators: two flags, four codepoints FLAGS = "\U0001f1eb\U0001f1ee\U0001f1f8\U0001f1ea" # base + variation selector 16 (render as emoji) VARIATION = "\u2764\ufe0f" def _is_mark(ch: str) -> bool: return unicodedata.category(ch) in {"Mn", "Mc", "Me"} # --- the failure a plain slice makes --------------------------------------- def test_a_plain_slice_orphans_a_combining_mark(): """The bug being prevented, stated as a fact about str slicing.""" assert ACCENT[:4] == "cafe" # accent silently dropped assert _is_mark(ACCENT[4:][0]) # tail begins with a bare accent # --- heads ------------------------------------------------------------------ def test_head_drops_the_base_rather_than_orphaning_its_mark(): assert safe_head(ACCENT, 4) == "caf" def test_head_keeps_a_whole_cluster_when_it_fits(): assert safe_head(ACCENT, 5) == ACCENT def test_head_does_not_split_a_zwj_sequence(): for n in range(1, len(ZWJ_FAMILY)): out = safe_head(ZWJ_FAMILY, n) assert not out.endswith("\u200d"), n assert out in ("", ZWJ_FAMILY[: len(out)]) assert safe_head(ZWJ_FAMILY, len(ZWJ_FAMILY) - 1) == "" def test_head_does_not_split_a_flag(): # One flag is two codepoints; cutting at 3 would leave one and a half. assert safe_head(FLAGS, 3) == FLAGS[:2] assert safe_head(FLAGS, 2) == FLAGS[:2] def test_head_does_not_orphan_a_variation_selector(): assert safe_head(VARIATION, 1) == "" assert safe_head(VARIATION, 2) == VARIATION # --- tails ------------------------------------------------------------------ def test_tail_does_not_begin_with_a_combining_mark(): out = safe_tail(ACCENT, 1) assert out == "" out = safe_tail(ACCENT, 2) assert out == "e\u0301" assert not _is_mark(out[0]) def test_tail_does_not_begin_mid_zwj_sequence(): for n in range(1, len(ZWJ_FAMILY)): out = safe_tail(ZWJ_FAMILY, n) assert not out.startswith("\u200d"), n def test_tail_does_not_begin_mid_flag(): assert safe_tail(FLAGS, 3) == FLAGS[2:] # --- invariants ------------------------------------------------------------- @pytest.mark.parametrize("text", [ACCENT, ZWJ_FAMILY, FLAGS, VARIATION, "plain ascii", "", "a"]) def test_neither_cut_ever_exceeds_its_budget(text): """n is a ceiling. A cut that grew to keep a cluster whole would let input the caller does not control blow past a budget it set.""" for n in range(0, len(text) + 3): assert len(safe_head(text, n)) <= max(n, 0) assert len(safe_tail(text, n)) <= max(n, 0) @pytest.mark.parametrize("text", [ACCENT, ZWJ_FAMILY, FLAGS, VARIATION]) def test_both_cuts_are_prefixes_and_suffixes_of_the_input(text): """Nothing is invented or reordered -- these only ever shorten.""" for n in range(0, len(text) + 3): assert text.startswith(safe_head(text, n)) assert text.endswith(safe_tail(text, n)) def test_ascii_is_untouched(): """The common case must not pay for the rare one.""" s = "the quick brown fox" for n in range(0, len(s) + 3): assert safe_head(s, n) == s[:n] if n <= len(s) else safe_head(s, n) == s if 0 < n <= len(s): assert safe_tail(s, n) == s[-n:] def test_a_full_length_request_returns_the_whole_string(): for text in (ACCENT, ZWJ_FAMILY, FLAGS, VARIATION): assert safe_head(text, len(text)) == text assert safe_tail(text, len(text)) == text