fix: pin vendored Cursor-skill text engine to the service copy

The engine vendored into skills/clean-user-facing-text/ had silently
drifted behind service/scripts/text_unicode.py: it still blanket-
stripped legitimate RTL directional marks and isolates (corrupting
mixed RTL/LTR prose) and stripped emoji variation selectors after
arrow and symbol bases, because it never received the preservable-bidi
and emoji-base updates. Nothing in the suite compared the two copies.

Replace the vendored engine with an exact copy of the service one and
add a byte-equality test so any future engine change must land in both
files in the same commit, plus a behavioural regression test for the
RTL and arrow cases through the vendored CLI. The CLI wrappers remain
deliberately different (text-only skill: no stylometry, no
--strip-bidi).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Pino De Francesco
2026-08-16 11:51:35 +01:00
co-authored by Claude Fable 5
parent fcebf53358
commit 519c6db9d6
3 changed files with 170 additions and 11 deletions
+1
View File
@@ -671,6 +671,7 @@ make smoke # quick CLI smoke on fixtures
### Unreleased
- **Fix vendored Cursor-skill engine drift and pin it byte-for-byte**: the text engine vendored into `skills/clean-user-facing-text/` had silently fallen behind the service copy, so the Cursor skill still blanket-stripped legitimate RTL directional marks and isolates (corrupting mixed RTL/LTR prose that isolates an LTR run with `U+2066``U+2069` and `U+200F`) and stripped emoji variation selectors after arrow and symbol bases (visibly changing sequences like `U+2194 U+FE0F`, ↔️ as one glyph). The vendored `text_unicode.py` is now an exact copy of the service engine and a new test pins the two files byte-for-byte, so future engine changes cannot land in one copy only. The CLI wrappers still deliberately differ (text-only skill: no stylometry, no `--strip-bidi`)
- **Fix `inspect` missing Layer A carriers in markdown/HTML**: `inspect_container` never scanned the document body, so a `.md` or `.html` file holding invisible Unicode came back `suspicious: false` while `clean_container` went on to strip it — the same bytes saved as `.txt` were correctly flagged. The scan now runs for exactly the formats `clean_container` scrubs, so inspect predicts clean. Container reports gain `suspicious_total` (the same key `TextInspectReport` uses, so the HTTP `suspicious` flag and the `inspect_file` exit code pick it up) and `layer_a_hits`
### [v0.5.0](https://github.com/guillaumemeyer/watermarks-remover/releases/tag/v0.5.0) — service & Docker distribution, HTTP API, and verification harnesses
@@ -5,6 +5,7 @@ from __future__ import annotations
import unicodedata
from collections import Counter
from dataclasses import dataclass, field
from difflib import SequenceMatcher
# Format / invisible controls commonly used for steganography or broken pastes.
STRIP_CODEPOINTS: frozenset[int] = frozenset(
@@ -185,6 +186,21 @@ _BIDI_CPS: frozenset[int] = frozenset(
}
)
# Directional marks and isolates are legitimate in mixed RTL/LTR prose. Inspect
# them, but preserve them during the default clean. Embeddings and overrides
# remain destructive by default because they can reorder unrelated spans.
_PRESERVABLE_BIDI_CPS: frozenset[int] = frozenset(
{
0x061C,
0x200E,
0x200F,
0x2066,
0x2067,
0x2068,
0x2069,
}
)
# Zero-width family (common edit-based carriers)
_ZW_FAMILY: frozenset[int] = frozenset(
{0x200B, 0x200C, 0x200D, 0x2060, 0xFEFF, 0x180E}
@@ -239,6 +255,8 @@ def _is_emoji_base(cp: int) -> bool:
"""Return True for characters that can start or continue an emoji sequence."""
if 0x1F000 <= cp <= 0x1FAFF:
return True
if 0x2190 <= cp <= 0x25FF: # arrows, technical symbols, enclosed symbols
return True
if 0x2600 <= cp <= 0x27BF: # misc symbols / dingbats / arrows
return True
if 0x2B00 <= cp <= 0x2BFF: # misc symbols and arrows
@@ -268,9 +286,71 @@ _HANGUL_FILLERS: frozenset[int] = frozenset({0x115F, 0x1160})
_SCRIPT_GLUE: frozenset[int] = _MONGOLIAN_FVS | _KHMER_VOWELS | _HANGUL_FILLERS
def _is_joining_letter(cp: int) -> bool:
"""Non-ASCII letter/mark — the neighbour that makes a joiner orthographic."""
return cp > 0x7F and unicodedata.category(chr(cp))[0] in ("L", "M")
def _joining_script(cp: int) -> str | None:
"""Return a broad script group where ZWJ/ZWNJ can be orthographic."""
for start, end, name in (
(0x0600, 0x08FF, "arabic"),
(0x0900, 0x0DFF, "indic"),
(0x0F00, 0x109F, "south-asian"),
(0x1780, 0x17FF, "khmer"),
(0x1800, 0x18AF, "mongolian"),
):
if start <= cp <= end and unicodedata.category(chr(cp))[0] in ("L", "M"):
return name
return None
def _is_cjk_ideograph(cp: int) -> bool:
return (
0x3400 <= cp <= 0x4DBF
or 0x4E00 <= cp <= 0x9FFF
or 0xF900 <= cp <= 0xFAFF
or 0x20000 <= cp <= 0x323AF
)
def _is_mongolian_base(cp: int) -> bool:
return 0x1800 <= cp <= 0x18AF
def _is_variation_selector(cp: int) -> bool:
return cp in _VS_SUPPLEMENT or 0xFE00 <= cp <= 0xFE0F or 0x180B <= cp <= 0x180D
def _valid_flag_tag_indices(text: str) -> set[int]:
"""Indices in complete subdivision-flag tag sequences."""
valid: set[int] = set()
i = 0
while i < len(text):
if ord(text[i]) != 0x1F3F4: # waving black flag
i += 1
continue
j = i + 1
while j < len(text) and 0xE0020 <= ord(text[j]) <= 0xE007E:
j += 1
if j > i + 1 and j < len(text) and ord(text[j]) == 0xE007F:
valid.update(range(i + 1, j + 1))
i = j + 1
else:
i += 1
return valid
def _valid_bidi_embedding_indices(text: str) -> set[int]:
"""Indices belonging to complete LRE/RLE ... PDF pairs, excluding overrides."""
valid: set[int] = set()
stack: list[tuple[int, int]] = []
for index, char in enumerate(text):
cp = ord(char)
if cp in (0x202A, 0x202B, 0x202D, 0x202E):
stack.append((cp, index))
elif cp == 0x202C:
if not stack:
continue
opener, opener_index = stack.pop()
if opener in (0x202A, 0x202B):
valid.update((opener_index, index))
return valid
def _is_mongolian_letter(cp: int) -> bool:
@@ -294,6 +374,7 @@ def _is_glue(cp: int) -> bool:
or same-script filler/selector (Mongolian FVS, Khmer vowel, Hangul filler)."""
return (
_is_emoji_glue(cp)
or _is_variation_selector(cp)
or cp in _SCRIPT_JOINERS
or cp in _TAG_RANGE
or cp in _SCRIPT_GLUE
@@ -303,10 +384,15 @@ def _is_glue(cp: int) -> bool:
def _decide(
ch: str,
prev_kept: str | None,
prev_input: str | None,
next_input: str | None,
*,
valid_flag_tag: bool,
valid_bidi_embedding: bool,
normalize_spaces: bool,
treat_confusables: bool,
strip_emoji_glue: bool,
strip_bidi: bool,
) -> tuple[str, str, str | None]:
"""Classify one input char for both inspect and clean.
@@ -315,13 +401,36 @@ def _decide(
kind is the inspect classification (None when not suspicious).
"""
cp = ord(ch)
if valid_bidi_embedding and not strip_bidi:
return ("keep", ch, None)
if cp in _PRESERVABLE_BIDI_CPS and not strip_bidi:
return ("keep", ch, None)
if prev_input is not None and not strip_emoji_glue:
prev_cp = ord(prev_input)
if cp in _VS_SUPPLEMENT and _is_cjk_ideograph(prev_cp):
return ("keep", ch, None)
if 0x180B <= cp <= 0x180D and _is_mongolian_base(prev_cp):
return ("keep", ch, None)
if 0xFE00 <= cp <= 0xFE0D and _is_cjk_ideograph(prev_cp):
return ("keep", ch, None)
if _is_emoji_glue(cp) and not strip_emoji_glue:
if prev_kept is not None and _is_emoji_base(ord(prev_kept)):
if cp in (0xFE0E, 0xFE0F) and prev_input is not None and _is_emoji_base(ord(prev_input)):
return ("keep", ch, None)
if (
cp == 0x200D
and prev_kept is not None
and next_input is not None
and _is_emoji_base(ord(prev_kept))
and _is_emoji_base(ord(next_input))
):
return ("keep", ch, None)
if not strip_emoji_glue:
if cp in _SCRIPT_JOINERS and prev_kept is not None and _is_joining_letter(ord(prev_kept)):
return ("keep", ch, None)
if cp in _TAG_RANGE and prev_kept is not None and _is_emoji_base(ord(prev_kept)):
if cp in _SCRIPT_JOINERS and prev_input is not None and next_input is not None:
prev_script = _joining_script(ord(prev_input))
next_script = _joining_script(ord(next_input))
if prev_script is not None and prev_script == next_script:
return ("keep", ch, None)
if cp in _TAG_RANGE and valid_flag_tag:
return ("keep", ch, None)
if cp in _MONGOLIAN_FVS and prev_kept is not None and _is_mongolian_letter(ord(prev_kept)):
return ("keep", ch, None)
@@ -398,13 +507,20 @@ def inspect_text(
) -> TextInspectReport:
buckets: dict[tuple[int, str], list[int]] = {}
prev_kept: str | None = None
valid_flag_tags = _valid_flag_tag_indices(text)
valid_bidi_embeddings = _valid_bidi_embedding_indices(text)
for i, ch in enumerate(text):
action, out_char, kind = _decide(
ch,
prev_kept,
text[i - 1] if i > 0 else None,
text[i + 1] if i + 1 < len(text) else None,
valid_flag_tag=i in valid_flag_tags,
valid_bidi_embedding=i in valid_bidi_embeddings,
normalize_spaces=True,
treat_confusables=aggressive,
strip_emoji_glue=strip_emoji_glue,
strip_bidi=True,
)
if kind is None:
# Kept; glue (emoji/script joiner/tag) does not advance the
@@ -438,7 +554,7 @@ def inspect_text(
"Layer A only: invisible/format Unicode and space homoglyphs (edit-based carriers).",
"Statistical (token-sampling) watermarks are not detectable here; use Layer B rewrite.",
"Inspect kinds: strip, bidi, tag_chars, variation_selector, zwj_family, private_use, space, confusable, other_cf.",
"Load-bearing invisibles are preserved by default: emoji glue (ZWJ/VS after an emoji base), script joiners (ZWNJ/ZWJ inside complex scripts), flag tag chars, same-script fillers/selectors (Mongolian FVS, Khmer inherent vowels, Hangul jamo fillers), and orthographic Arabic/Syriac Cf marks. Use --strip-emoji-glue for paranoid mode (strips them all).",
"Load-bearing invisibles are preserved by default during cleaning: emoji glue, CJK/Mongolian variation selectors, script joiners, complete flag tag sequences, same-script fillers/selectors (Mongolian FVS, Khmer inherent vowels, Hangul jamo fillers), RTL directional marks/paired embeddings, and orthographic Arabic/Syriac Cf marks. Inspection still reports bidi controls. Use explicit strip flags only after review.",
]
if not hits:
notes.append(
@@ -455,20 +571,28 @@ def clean_text(
aggressive_homoglyphs: bool = False,
normalize_spaces: bool = True,
strip_emoji_glue: bool = False,
strip_bidi: bool = False,
) -> tuple[str, dict]:
"""Return cleaned text and a stats dict."""
removed: Counter[str] = Counter()
replaced: Counter[str] = Counter()
out_chars: list[str] = []
prev_kept: str | None = None
valid_flag_tags = _valid_flag_tag_indices(text)
valid_bidi_embeddings = _valid_bidi_embedding_indices(text)
for ch in text:
for i, ch in enumerate(text):
action, out_char, _kind = _decide(
ch,
prev_kept,
text[i - 1] if i > 0 else None,
text[i + 1] if i + 1 < len(text) else None,
valid_flag_tag=i in valid_flag_tags,
valid_bidi_embedding=i in valid_bidi_embeddings,
normalize_spaces=normalize_spaces,
treat_confusables=aggressive_homoglyphs,
strip_emoji_glue=strip_emoji_glue,
strip_bidi=strip_bidi,
)
if action == "keep":
out_chars.append(out_char)
@@ -485,11 +609,20 @@ def clean_text(
# prev_kept unchanged
result = "".join(out_chars)
nfkc_changed = False
if nfkc:
before = result
result = unicodedata.normalize("NFKC", result)
if result != before:
replaced["NFKC_normalize"] += abs(len(before) - len(result)) or 1
nfkc_changed = True
changed_inputs = sum(
end - start
for operation, start, end, _new_start, _new_end in SequenceMatcher(
None, before, result, autojunk=False
).get_opcodes()
if operation != "equal"
)
replaced["NFKC_normalize"] += changed_inputs or 1
# Collapse runs of spaces only if we introduced space replacements? Keep conservative: no.
@@ -499,7 +632,8 @@ def clean_text(
"removed": dict(removed),
"replaced": dict(replaced),
"removed_count": sum(removed.values()),
"replaced_count": sum(v for k, v in replaced.items() if k != "NFKC_normalize"),
"replaced_count": sum(replaced.values()),
"nfkc_changed": nfkc_changed,
}
return result, stats
+24
View File
@@ -74,3 +74,27 @@ def test_installer_force_creates_backup_and_replaces(tmp_path):
assert len(backups) == 1
assert (backups[0] / "old").read_text(encoding="utf-8") == "old"
assert (destination / "SKILL.md").is_file()
def test_vendored_text_unicode_is_identical_to_service_engine():
# The Layer A engine is vendored byte-for-byte; only the CLI wrappers
# (clean_text.py, inspect_text.py, common.py) may differ. Any engine
# change must be applied to both copies in the same commit.
service = (ROOT / "service" / "scripts" / "text_unicode.py").read_bytes()
vendored = (SKILL / "scripts" / "text_unicode.py").read_bytes()
assert service == vendored
def test_lightweight_preserves_legitimate_bidi_and_emoji_glue():
# Regression for engine drift: the vendored copy once blanket-stripped
# RTL isolates/marks and VS16 after arrows, corrupting legitimate text.
for raw in ("السعر 123 USD", "Move ↔️"):
result = subprocess.run(
[sys.executable, str(SKILL / "scripts" / "clean_text.py"), "-"],
input=raw,
text=True,
encoding="utf-8",
capture_output=True,
check=True,
)
assert result.stdout.rstrip("\n") == raw