diff --git a/src/interscript/engine.py b/src/interscript/engine.py index 0d4afad..a7172a1 100644 --- a/src/interscript/engine.py +++ b/src/interscript/engine.py @@ -17,7 +17,7 @@ import re import unicodedata -from .expr import expr_to_literal, expr_to_regex, is_plain_string +from .expr import expr_max_length, expr_to_literal, expr_to_regex, is_plain_string class ExecutionError(ValueError): @@ -38,13 +38,18 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str] full = "(?<=" + expr_to_regex(sub["before"]) + ")" + full if sub.get("after"): full = full + "(?=" + expr_to_regex(sub["after"]) + ")" - indexed.append((pat, full, f"s{i}")) + key = expr_max_length(sub["pattern"]) + for guard in ("before", "after", "not_before", "not_after"): + if sub.get(guard): + key += expr_max_length(sub[guard]) + key += sub.get("priority", 0) + indexed.append((key, full, f"s{i}")) if is_plain_string(sub["pattern"]) and not sub.get("before") and not sub.get("after"): src = expr_to_literal(sub["pattern"]) if src.upper() != src: anchor_results[f"a{i}"] = expr_to_literal(sub["result"]) - indexed.append((re.escape(src.upper()), re.escape(src.upper()), f"a{i}")) - indexed.sort(key=lambda t: -len(t[0])) + indexed.append((len(src), re.escape(src.upper()), f"a{i}")) + indexed.sort(key=lambda t: -t[0]) combined = "|".join(f"(?P<{name}>{full})" for _, full, name in indexed) pattern = re.compile(combined) if indexed else re.compile(r"(?!)") diff --git a/src/interscript/expr.py b/src/interscript/expr.py index 7b21151..ddbc139 100644 --- a/src/interscript/expr.py +++ b/src/interscript/expr.py @@ -123,3 +123,23 @@ def expr_to_literal(expr: str) -> str: def is_plain_string(expr: str) -> bool: return bool(re.fullmatch(r'"(?:[^"\\]|\\.)*"', expr.strip())) + + +def expr_max_length(expr: str) -> int: + """The Ruby runtime's parallel-selection key: Rule::Sub#max_length = + from + before + after + not_before + not_after (+ priority), where a + zero-width stdlib alias (boundary, line_start, ...) counts 1.""" + total = 0 + for kind, value in _scan(expr, "expression"): + if kind == "lit": + total += len(value) + elif kind in ("cls", "range", "space", "boundary", "nwb", "anchor"): + total += 1 + elif kind == "alt": + total += max(len(a) for a in value.split("\x00")) + elif kind == "opt": + total += len(value) + elif kind == "grp": + total += expr_max_length(value) + # cat: concatenation marker + return total diff --git a/tests/test_engine.py b/tests/test_engine.py index ecdb104..2aa1c11 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -115,3 +115,23 @@ def test_engine_separate_and_title_case(): ) assert transliterate("tc", "hello world") == "Hello world" assert transliterate("tc", "hello world\nhello hello") == "Hello world\nHello hello" + + +def test_parallel_selection_matches_ruby_max_length(): + """Selection mirrors Ruby Rule::Sub#max_length: from + all guard + lengths (+priority), zero-width aliases counting 1. A longer later + rule (4) must beat an earlier boundary-guarded one (3) — the + alalc-ara hamza shape ("\u0623\u064e" vs boundary+"\u0623").""" + tree = parse_imp( + 'stage {\n parallel {\n' + ' sub boundary + "ab", "X"\n' + ' sub "abcd", "Y"\n }\n}\n' + ) + assert Engine(tree).transliterate("abcd") == "Y" + # Equal keys keep source order, like Ruby's index tiebreak. + tree2 = parse_imp( + 'stage {\n parallel {\n' + ' sub "ab", "X"\n' + ' sub "abc", "Y"\n }\n}\n' + ) + assert Engine(tree2).transliterate("abc") == "Y"