From d7243f3edfbd9390cfb0b416f005cd953a5742b7 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Thu, 1 Oct 2026 17:46:09 +0800 Subject: [PATCH] fix: boundary treats combining marks as word characters, like Ruby MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ruby's \b counts combining marks (Arabic diacritics) as word characters; Python's \w does not — Mn is not alphanumeric. At a hamza-carrier + kasra junction Ruby saw no boundary while Python did, so every word-final rule (X + boundary -> "'a") fired wrongly and doubled vowels: دَائِم became dā'aim instead of dā'im. The boundary now compiles to an explicit word/non-word transition over a word class that includes combining marks. Through the Ruby bridge, alalc-ara: 10 failures -> 4. --- src/interscript/expr.py | 12 +++++++++++- tests/test_engine.py | 17 +++++++++++++++++ 2 files changed, 28 insertions(+), 1 deletion(-) diff --git a/src/interscript/expr.py b/src/interscript/expr.py index ddbc139..78a133f 100644 --- a/src/interscript/expr.py +++ b/src/interscript/expr.py @@ -29,6 +29,16 @@ SPACE = re.escape(" ") +# Ruby's \b counts combining marks as word characters; Python's \w +# does not (Mn is not alphanumeric). At a hamza-carrier + kasra +# junction Ruby sees no boundary while Python does — word-final rules +# fired wrongly and doubled vowels. Express the boundary as an +# explicit word/non-word transition over a word class that includes +# combining marks (Mnemonic ranges: combining diacritics 0300-036F, +# Arabic diacritics 064B-065F, 0670, and Quranic annotation 06D6-06ED). +_WORD = r"[\w\u0300-\u036F\u064B-\u065F\u0670\u06D6-\u06ED]" +_BOUNDARY = "(?:(?<=" + _WORD + ")(?!" + _WORD + ")|(? str: return _UNESC.sub(lambda m: chr(int(m.group(1), 16)), s) @@ -95,7 +105,7 @@ def expr_to_regex(expr: str) -> str: elif kind == "space": parts.append(SPACE) elif kind == "boundary": - parts.append(r"\b") + parts.append(_BOUNDARY) elif kind == "nwb": parts.append(r"\B") elif kind == "grp": diff --git a/tests/test_engine.py b/tests/test_engine.py index 2aa1c11..6cbc0e9 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -135,3 +135,20 @@ def test_parallel_selection_matches_ruby_max_length(): ' sub "abc", "Y"\n }\n}\n' ) assert Engine(tree2).transliterate("abc") == "Y" + + +def test_boundary_treats_combining_marks_as_word_chars(): + """Ruby's \\b counts combining marks (Arabic diacritics) as word + characters — a word-final rule must not fire when a kasra follows + the hamza carrier (dā'im, not dā'aim).""" + tree = parse_imp( + 'stage {\n parallel {\n' + ' sub "ئ" + boundary, "\'a"\n' + ' sub "ئ", "\'"\n' + ' sub "ِ", "i"\n' + ' sub "d", "d"\n }\n}\n' + ) + # ئ + kasra: no boundary — the bare-ئ rule fires, not the final one. + assert Engine(tree).transliterate("dئِ") == "d'i" + # ئ at a true word end: the boundary rule fires. + assert Engine(tree).transliterate("dئ") == "d'a"