diff --git a/src/interscript/expr.py b/src/interscript/expr.py index ddbc139..78a133f 100644 --- a/src/interscript/expr.py +++ b/src/interscript/expr.py @@ -29,6 +29,16 @@ SPACE = re.escape(" ") +# Ruby's \b counts combining marks as word characters; Python's \w +# does not (Mn is not alphanumeric). At a hamza-carrier + kasra +# junction Ruby sees no boundary while Python does — word-final rules +# fired wrongly and doubled vowels. Express the boundary as an +# explicit word/non-word transition over a word class that includes +# combining marks (Mnemonic ranges: combining diacritics 0300-036F, +# Arabic diacritics 064B-065F, 0670, and Quranic annotation 06D6-06ED). +_WORD = r"[\w\u0300-\u036F\u064B-\u065F\u0670\u06D6-\u06ED]" +_BOUNDARY = "(?:(?<=" + _WORD + ")(?!" + _WORD + ")|(? str: return _UNESC.sub(lambda m: chr(int(m.group(1), 16)), s) @@ -95,7 +105,7 @@ def expr_to_regex(expr: str) -> str: elif kind == "space": parts.append(SPACE) elif kind == "boundary": - parts.append(r"\b") + parts.append(_BOUNDARY) elif kind == "nwb": parts.append(r"\B") elif kind == "grp": diff --git a/tests/test_engine.py b/tests/test_engine.py index 2aa1c11..6cbc0e9 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -135,3 +135,20 @@ def test_parallel_selection_matches_ruby_max_length(): ' sub "abc", "Y"\n }\n}\n' ) assert Engine(tree2).transliterate("abc") == "Y" + + +def test_boundary_treats_combining_marks_as_word_chars(): + """Ruby's \\b counts combining marks (Arabic diacritics) as word + characters — a word-final rule must not fire when a kasra follows + the hamza carrier (dā'im, not dā'aim).""" + tree = parse_imp( + 'stage {\n parallel {\n' + ' sub "ئ" + boundary, "\'a"\n' + ' sub "ئ", "\'"\n' + ' sub "ِ", "i"\n' + ' sub "d", "d"\n }\n}\n' + ) + # ئ + kasra: no boundary — the bare-ئ rule fires, not the final one. + assert Engine(tree).transliterate("dئِ") == "d'i" + # ئ at a true word end: the boundary rule fires. + assert Engine(tree).transliterate("dئ") == "d'a"