From 43faa58cecced0f7d66c7941bf28abfd712e8edc Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Thu, 1 Oct 2026 18:12:26 +0800 Subject: [PATCH] fix: compile not_before/not_after as guards; full expressions in any lists MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ruby's interpreter compiles not_before to a negative lookbehind and not_after to a negative lookahead (build_regexp). The Python engine counted them in the selection key but never matched against them, so rules like alalc-ara's hamza+fatḥa -> 'a' fired where Ruby declines them: mas'alah came out masalah, ta'ālīf came out tālīf. Three coupled fixes: - isc.py renders not_before/not_after constraints into parallel subs (they were dropped at rendering) and engine.py compiles them; the negative lookbehind distributes over top-level alternatives because Python re requires fixed-width lookbehinds where Onigmo does not. - any([...]) alternatives are full expressions, not quoted strings only. The list tokenizer kept just the quoted strings, silently dropping bare atoms (boundary) and alias references; alternatives now tokenize recursively (regex, literal, max_length). - _library_aliases substitutes alias references defined earlier in the same .iml file (var-kor: jamo = any([jamo_leading_cons, ...])), which the dropped-alternatives path used to hide. alalc-ara: 94/94. un-bul (imports var-kor) and bgnpcgn-deu load and pass again with the previously-dropped alternatives now resolved. --- src/interscript/engine.py | 10 ++-- src/interscript/expr.py | 98 ++++++++++++++++++++++++++++++++++----- src/interscript/isc.py | 38 ++++++++++++++- tests/test_engine.py | 60 ++++++++++++++++++++++++ tests/test_isc.py | 19 ++++++++ 5 files changed, 209 insertions(+), 16 deletions(-) diff --git a/src/interscript/engine.py b/src/interscript/engine.py index a7172a1..4e400ea 100644 --- a/src/interscript/engine.py +++ b/src/interscript/engine.py @@ -17,7 +17,7 @@ import re import unicodedata -from .expr import expr_max_length, expr_to_literal, expr_to_regex, is_plain_string +from .expr import expr_max_length, expr_neg_lookbehind, expr_to_literal, expr_to_regex, is_plain_string class ExecutionError(ValueError): @@ -36,6 +36,10 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str] full = pat if sub.get("before"): full = "(?<=" + expr_to_regex(sub["before"]) + ")" + full + if sub.get("not_before"): + full = expr_neg_lookbehind(sub["not_before"]) + full + if sub.get("not_after"): + full = full + "(?!" + expr_to_regex(sub["not_after"]) + ")" if sub.get("after"): full = full + "(?=" + expr_to_regex(sub["after"]) + ")" key = expr_max_length(sub["pattern"]) @@ -44,7 +48,7 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str] key += expr_max_length(sub[guard]) key += sub.get("priority", 0) indexed.append((key, full, f"s{i}")) - if is_plain_string(sub["pattern"]) and not sub.get("before") and not sub.get("after"): + if is_plain_string(sub["pattern"]) and not any(sub.get(g) for g in ("before", "after", "not_before", "not_after")): src = expr_to_literal(sub["pattern"]) if src.upper() != src: anchor_results[f"a{i}"] = expr_to_literal(sub["result"]) @@ -58,7 +62,7 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str] casing_map: dict[str, str] = {} upper_dst: dict[str, str] = {} for sub in subs: - if is_plain_string(sub["pattern"]) and not sub.get("before") and not sub.get("after"): + if is_plain_string(sub["pattern"]) and not any(sub.get(g) for g in ("before", "after", "not_before", "not_after")): src = expr_to_literal(sub["pattern"]) dst = expr_to_literal(sub["result"]) casing_map[src] = dst diff --git a/src/interscript/expr.py b/src/interscript/expr.py index 78a133f..90ab9dd 100644 --- a/src/interscript/expr.py +++ b/src/interscript/expr.py @@ -17,14 +17,63 @@ r'|any\(\s*"(?P(?:[^"\\]|\\.)*)"\s*\.\.\s*"(?P(?:[^"\\]|\\.)*)"\s*\)' r'|any\(\s*"(?P(?:[^"\\]|\\.)*)"\s*\)' r'|maybe\(\s*"(?P(?:[^"\\]|\\.)*)"\s*\)' - r'|any\(\s*\[(?P(?:[^\\\[\]]|\\.)*)\]\s*\)' r"|(?P\bspace\b)|(?P\bboundary\b)" r"|(?P\bnon_word_boundary\b)" r'|capture\(\s*(?P(?:[^()\\]|\\.|\([^()]*\))*)\s*\)' r"|(?P\bline_end\b)|(?P\bline_start\b)" r"|(?P\+)" ) -_LIST_SPLIT = re.compile(r'"((?:[^"\\]|\\.)*)"') +def _split_list(src: str) -> list[str]: + """Alternatives of any([...]) are full expressions — split on + commas outside quotes and brackets (boundary + "ab" keeps its + boundary atom; any([a, b]) nests).""" + parts, buf, quote, depth = [], [], None, 0 + for ch in src: + if quote: + buf.append(ch) + if ch == quote: + quote = None + elif ch in "\"'": + buf.append(ch) + quote = ch + elif ch == "[": + depth += 1 + buf.append(ch) + elif ch == "]": + depth -= 1 + buf.append(ch) + elif ch == "," and depth == 0: + parts.append("".join(buf)) + buf = [] + else: + buf.append(ch) + parts.append("".join(buf)) + return [p.strip() for p in parts if p.strip()] + + +def _read_bracketed(expr: str, pos: int) -> tuple[str, int]: + """With pos at the '[' of any([: return the inner content and the + position after the matching ']' (quote- and depth-aware).""" + depth, i, n = 0, pos, len(expr) + while i < n: + c = expr[i] + if c == '"': + i += 1 + while i < n: + if expr[i] == "\\": + i += 2 + continue + if expr[i] == '"': + break + i += 1 + elif c == "[": + depth += 1 + elif c == "]": + depth -= 1 + if depth == 0: + return expr[pos + 1 : i], i + 1 + i += 1 + raise ValueError(f"unterminated any([ in {expr!r}") _UNESC = re.compile(r"\\u([0-9a-fA-F]{4})") SPACE = re.escape(" ") @@ -44,14 +93,30 @@ def _unesc(s: str) -> str: return _UNESC.sub(lambda m: chr(int(m.group(1), 16)), s) +_ANY_LIST = re.compile(r"any\(\s*\[") + + def _scan(expr: str, want: str): """Tokenize an expression into (kind, value); raises on gaps.""" out: list[tuple[str, str]] = [] pos = 0 - for m in _TOKEN.finditer(expr): - gap = expr[pos : m.start()] - if gap.strip(): - raise ValueError(f"cannot parse {want} near {gap.strip()!r} in {expr!r}") + while pos < len(expr): + if expr[pos].isspace(): + pos += 1 + continue + if _ANY_LIST.match(expr, pos): + bracket = expr.find("[", pos) + inner, pos = _read_bracketed(expr, bracket) + while pos < len(expr) and expr[pos].isspace(): + pos += 1 + if pos >= len(expr) or expr[pos] != ")": + raise ValueError(f"expected ')' closing any([ in {expr!r}") + pos += 1 + out.append(("alt", "\x00".join(_split_list(inner)))) + continue + m = _TOKEN.match(expr, pos) + if not m: + raise ValueError(f"cannot parse {want} near {expr[pos : pos + 20]!r} in {expr!r}") pos = m.end() g = m.groupdict() if g["lit"] is not None: @@ -62,9 +127,6 @@ def _scan(expr: str, want: str): out.append(("cls", _unesc(g["cls"]))) elif g["opt"] is not None: out.append(("opt", _unesc(g["opt"]))) - elif g["lst"] is not None: - alts = [_unesc(x) for x in _LIST_SPLIT.findall(g["lst"])] - out.append(("alt", "\x00".join(alts))) elif g["space"] is not None: out.append(("space", " ")) elif g["boundary"] is not None: @@ -87,6 +149,18 @@ def _scan(expr: str, want: str): return out +def expr_neg_lookbehind(expr: str) -> str: + """Negative lookbehind over an expression. Python re requires + fixed-width lookbehinds (Ruby's Onigmo does not), so a top-level + alternation is distributed: (? str: parts = [] for kind, value in _scan(expr, "expression"): @@ -99,7 +173,7 @@ def expr_to_regex(expr: str) -> str: parts.append("[" + re.escape(lo) + "-" + re.escape(hi) + "]") elif kind == "alt": alts = value.split("\x00") - parts.append("(?:" + "|".join(re.escape(a) for a in alts) + ")") + parts.append("(?:" + "|".join(expr_to_regex(a) for a in alts) + ")") elif kind == "opt": parts.append("(?:" + re.escape(value) + ")?") elif kind == "space": @@ -123,7 +197,7 @@ def expr_to_literal(expr: str) -> str: elif kind == "cls": parts.append(value[0]) # deterministic: first alternative elif kind == "alt": - parts.append(value.split("\x00")[0]) + parts.append(expr_to_literal(value.split("\x00")[0])) elif kind == "space": parts.append(" ") elif kind in ("boundary", "anchor"): @@ -146,7 +220,7 @@ def expr_max_length(expr: str) -> int: elif kind in ("cls", "range", "space", "boundary", "nwb", "anchor"): total += 1 elif kind == "alt": - total += max(len(a) for a in value.split("\x00")) + total += max(expr_max_length(a) for a in value.split("\x00")) elif kind == "opt": total += len(value) elif kind == "grp": diff --git a/src/interscript/isc.py b/src/interscript/isc.py index 6b693e2..05137f0 100644 --- a/src/interscript/isc.py +++ b/src/interscript/isc.py @@ -832,8 +832,12 @@ def _stage_tree( for constraint in rule["constraints"]: if constraint["kind"] == "before": sub["before"] = _render_item(constraint["item"], aliases) + elif constraint["kind"] == "not_before": + sub["not_before"] = _render_item(constraint["item"], aliases) elif constraint["kind"] == "after": sub["after"] = _render_item(constraint["item"], aliases) + elif constraint["kind"] == "not_after": + sub["not_after"] = _render_item(constraint["item"], aliases) subs.append(sub) if capture_rules: # Capture-bearing rules degrade to ordered substitutions ahead @@ -950,11 +954,43 @@ def _library_aliases(name: str, libs_dir: Path) -> dict[str, str]: value = m.group(2) if "'" in value and '"' not in value: value = value.replace("'", '"') - aliases[m.group(1)] = value + # Library aliases may reference aliases defined earlier + # in the same file (var-kor: jamo = any([jamo_leading_cons, + # ...])); substitute them, outside string literals only. + aliases[m.group(1)] = _subst_alias_refs(value, aliases) _LIB_CACHE[name] = aliases return aliases +def _subst_alias_refs(value: str, resolved: dict[str, str]) -> str: + out, i, n = [], 0, len(value) + while i < n: + c = value[i] + if c == '"': + j = i + 1 + while j < n: + if value[j] == "\\": + j += 2 + continue + if value[j] == '"': + j += 1 + break + j += 1 + out.append(value[i:j]) + i = j + elif c.isascii() and (c.isalnum() or c == "_"): + j = i + while j < n and (value[j].isascii() and (value[j].isalnum() or value[j] in "_-")): + j += 1 + word = value[i:j] + out.append(resolved.get(word, word)) + i = j + else: + out.append(c) + i += 1 + return "".join(out) + + def isc_to_tree(source: str, filename: str | None = None, on_unsupported: str = "raise") -> dict: """Parse .isc source and convert to the engine's tree shape. diff --git a/tests/test_engine.py b/tests/test_engine.py index 6cbc0e9..314750e 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -152,3 +152,63 @@ def test_boundary_treats_combining_marks_as_word_chars(): assert Engine(tree).transliterate("dئِ") == "d'i" # ئ at a true word end: the boundary rule fires. assert Engine(tree).transliterate("dئ") == "d'a" + + +@pytest.mark.skipif(not MAPS.is_dir(), reason="interscript maps repo not present") +def test_not_guards_are_match_constraints_not_sort_keys_only(): + """Ruby compiles not_before/not_after as real guards (negative + lookbehind/lookahead — interpreter#build_regexp). The alalc-ara + shape: mas'alah keeps the hamza mark only because أ+fatḥa -> 'a' + declines when ة/ل FOLLOWS (not_after is following context).""" + import tempfile, os + from interscript import add_load_path, transliterate + + d = tempfile.mkdtemp() + open(os.path.join(d, "ng.isc"), "w").write( + "system \"ng\" {\n" + " stage main {\n" + " parallel {\n" + " sub {\n" + " from \"xy\"\n" + " to \"A\"\n" + " not_after any(\"ab\")\n" + " }\n" + " sub {\n" + " from \"zw\"\n" + " to \"B\"\n" + " not_before any(\"pq\")\n" + " }\n" + " sub \"x\" \"Q\"\n" + " sub \"z\" \"Z\"\n" + " }\n" + " }\n" + "}\n" + ) + add_load_path(d) + # not_after: "xy" followed by a/b declines -> bare x rule fires. + assert transliterate("ng", "xya") == "Qya" + assert transliterate("ng", "xyc") == "Ac" + # not_before: "zw" preceded by p/q declines. + assert transliterate("ng", "pzw") == "pZw" + assert transliterate("ng", "czw") == "cB" + + +def test_any_list_alternatives_are_full_expressions(): + """any([...]) alternatives are full items (Ruby: Any of Items, each + may be a Group). The list tokenizer kept only quoted strings, so + bare atoms like boundary were silently dropped and not_before + guards lost their boundary branch — alalc-ara word-initial آ then + took the medial "’ā" rule instead of "ā".""" + tree = parse_imp( + 'stage {\n parallel {\n' + ' sub any([boundary + "ab", "q"]), "X"\n' + ' sub "b", "Y"\n' + ' }\n}\n' + ) + e = Engine(tree) + # boundary-guarded alternative fires at word start. + assert e.transliterate("ab") == "X" + assert e.transliterate("q") == "X" + # no boundary before "ab" -> the alternative declines, bare b fires. + assert e.transliterate("xab") == "xaY" + assert e.transliterate("cab") == "caY" diff --git a/tests/test_isc.py b/tests/test_isc.py index 22897a2..74a416c 100644 --- a/tests/test_isc.py +++ b/tests/test_isc.py @@ -2,6 +2,9 @@ from __future__ import annotations +import os +from pathlib import Path + import pytest from interscript.isc import IscParseError, UnsupportedConstruct, isc_to_tree @@ -267,3 +270,19 @@ def test_imported_alias_resolves(): subst = tree["stages"][0]["children"][0] assert subst["pattern"] == "404" assert subst["result"] == "500" + + +@pytest.mark.skipif( + not Path(os.environ.get("INTERSCRIPT_MAPS_PATH", Path(__file__).parent.parent.parent / "maps" / "maps")).is_dir(), + reason="interscript maps repo not present", +) +def test_library_aliases_resolve_intra_library_references(): + """var-kor defines jamo in terms of other library aliases; the + harvested value must have those substituted (they were previously + dropped silently by the any-list tokenizer).""" + from interscript.isc import _library_aliases + + libs = Path(os.environ.get("INTERSCRIPT_MAPS_PATH", Path(__file__).parent.parent.parent / "maps" / "maps")).parent / "libs" + al = _library_aliases("var-kor", libs) + assert "jamo_leading_cons" not in al["jamo"] + assert "\\u1100" in al["jamo"] or "ᄀ" in al["jamo"]