Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 9 additions & 4 deletions src/interscript/engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@
import re
import unicodedata

from .expr import expr_to_literal, expr_to_regex, is_plain_string
from .expr import expr_max_length, expr_to_literal, expr_to_regex, is_plain_string


class ExecutionError(ValueError):
Expand All @@ -38,13 +38,18 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str]
full = "(?<=" + expr_to_regex(sub["before"]) + ")" + full
if sub.get("after"):
full = full + "(?=" + expr_to_regex(sub["after"]) + ")"
indexed.append((pat, full, f"s{i}"))
key = expr_max_length(sub["pattern"])
for guard in ("before", "after", "not_before", "not_after"):
if sub.get(guard):
key += expr_max_length(sub[guard])
key += sub.get("priority", 0)
indexed.append((key, full, f"s{i}"))
if is_plain_string(sub["pattern"]) and not sub.get("before") and not sub.get("after"):
src = expr_to_literal(sub["pattern"])
if src.upper() != src:
anchor_results[f"a{i}"] = expr_to_literal(sub["result"])
indexed.append((re.escape(src.upper()), re.escape(src.upper()), f"a{i}"))
indexed.sort(key=lambda t: -len(t[0]))
indexed.append((len(src), re.escape(src.upper()), f"a{i}"))
indexed.sort(key=lambda t: -t[0])
combined = "|".join(f"(?P<{name}>{full})" for _, full, name in indexed)
pattern = re.compile(combined) if indexed else re.compile(r"(?!)")

Expand Down
20 changes: 20 additions & 0 deletions src/interscript/expr.py
Original file line number Diff line number Diff line change
Expand Up @@ -123,3 +123,23 @@ def expr_to_literal(expr: str) -> str:

def is_plain_string(expr: str) -> bool:
return bool(re.fullmatch(r'"(?:[^"\\]|\\.)*"', expr.strip()))


def expr_max_length(expr: str) -> int:
"""The Ruby runtime's parallel-selection key: Rule::Sub#max_length =
from + before + after + not_before + not_after (+ priority), where a
zero-width stdlib alias (boundary, line_start, ...) counts 1."""
total = 0
for kind, value in _scan(expr, "expression"):
if kind == "lit":
total += len(value)
elif kind in ("cls", "range", "space", "boundary", "nwb", "anchor"):
total += 1
elif kind == "alt":
total += max(len(a) for a in value.split("\x00"))
elif kind == "opt":
total += len(value)
elif kind == "grp":
total += expr_max_length(value)
# cat: concatenation marker
return total
20 changes: 20 additions & 0 deletions tests/test_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -115,3 +115,23 @@ def test_engine_separate_and_title_case():
)
assert transliterate("tc", "hello world") == "Hello world"
assert transliterate("tc", "hello world\nhello hello") == "Hello world\nHello hello"


def test_parallel_selection_matches_ruby_max_length():
"""Selection mirrors Ruby Rule::Sub#max_length: from + all guard
lengths (+priority), zero-width aliases counting 1. A longer later
rule (4) must beat an earlier boundary-guarded one (3) — the
alalc-ara hamza shape ("\u0623\u064e" vs boundary+"\u0623")."""
tree = parse_imp(
'stage {\n parallel {\n'
' sub boundary + "ab", "X"\n'
' sub "abcd", "Y"\n }\n}\n'
)
assert Engine(tree).transliterate("abcd") == "Y"
# Equal keys keep source order, like Ruby's index tiebreak.
tree2 = parse_imp(
'stage {\n parallel {\n'
' sub "ab", "X"\n'
' sub "abc", "Y"\n }\n}\n'
)
assert Engine(tree2).transliterate("abc") == "Y"
Loading