diff --git a/src/interscript/engine.py b/src/interscript/engine.py index 4e400ea..5c4c4b4 100644 --- a/src/interscript/engine.py +++ b/src/interscript/engine.py @@ -17,7 +17,7 @@ import re import unicodedata -from .expr import expr_max_length, expr_neg_lookbehind, expr_to_literal, expr_to_regex, is_plain_string +from .expr import expr_lookbehind, expr_max_length, expr_neg_lookbehind, expr_to_literal, expr_to_regex, is_plain_string class ExecutionError(ValueError): @@ -35,7 +35,7 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str] pat = expr_to_regex(sub["pattern"]) full = pat if sub.get("before"): - full = "(?<=" + expr_to_regex(sub["before"]) + ")" + full + full = expr_lookbehind(sub["before"]) + full if sub.get("not_before"): full = expr_neg_lookbehind(sub["not_before"]) + full if sub.get("not_after"): diff --git a/src/interscript/expr.py b/src/interscript/expr.py index 90ab9dd..fdfd689 100644 --- a/src/interscript/expr.py +++ b/src/interscript/expr.py @@ -16,7 +16,6 @@ r'"(?P(?:[^"\\]|\\.)*)"' r'|any\(\s*"(?P(?:[^"\\]|\\.)*)"\s*\.\.\s*"(?P(?:[^"\\]|\\.)*)"\s*\)' r'|any\(\s*"(?P(?:[^"\\]|\\.)*)"\s*\)' - r'|maybe\(\s*"(?P(?:[^"\\]|\\.)*)"\s*\)' r"|(?P\bspace\b)|(?P\bboundary\b)" r"|(?P\bnon_word_boundary\b)" r'|capture\(\s*(?P(?:[^()\\]|\\.|\([^()]*\))*)\s*\)' @@ -94,6 +93,32 @@ def _unesc(s: str) -> str: _ANY_LIST = re.compile(r"any\(\s*\[") +_MAYBE = re.compile(r"maybe\(\s*") + + +def _read_parenthesized(expr: str, pos: int) -> tuple[str, int]: + """With pos at '(': return the inner content and the position after + the matching ')' (quote-, bracket- and depth-aware).""" + depth, i, n = 0, pos, len(expr) + while i < n: + c = expr[i] + if c == '"': + i += 1 + while i < n: + if expr[i] == "\\": + i += 2 + continue + if expr[i] == '"': + break + i += 1 + elif c == "(": + depth += 1 + elif c == ")": + depth -= 1 + if depth == 0: + return expr[pos + 1 : i], i + 1 + i += 1 + raise ValueError(f"unterminated parenthesis in {expr!r}") def _scan(expr: str, want: str): @@ -104,6 +129,11 @@ def _scan(expr: str, want: str): if expr[pos].isspace(): pos += 1 continue + if _MAYBE.match(expr, pos): + paren = expr.find("(", pos) + inner, pos = _read_parenthesized(expr, paren) + out.append(("opt", inner)) + continue if _ANY_LIST.match(expr, pos): bracket = expr.find("[", pos) inner, pos = _read_bracketed(expr, bracket) @@ -125,8 +155,6 @@ def _scan(expr: str, want: str): out.append(("range", _unesc(g["rlo"]) + "\x00" + _unesc(g["rhi"]))) elif g["cls"] is not None: out.append(("cls", _unesc(g["cls"]))) - elif g["opt"] is not None: - out.append(("opt", _unesc(g["opt"]))) elif g["space"] is not None: out.append(("space", " ")) elif g["boundary"] is not None: @@ -149,16 +177,50 @@ def _scan(expr: str, want: str): return out +def _lb_branches(expr: str) -> list[str]: + """Expand an expression into fixed-width regex branches by cross + product over alternations and optionals. Python re requires + fixed-width lookbehinds where Ruby's Onigmo does not, so guards + distribute over their branches instead.""" + branches = [""] + for kind, value in _scan(expr, "lookbehind"): + if kind == "cat": + continue + if kind == "alt": + expanded = [b for a in value.split("\x00") for b in _lb_branches(a)] + elif kind == "opt": + expanded = _lb_branches(value) + [""] + elif kind == "grp": + expanded = ["(" + expr_to_regex(value) + ")"] + elif kind == "lit": + expanded = [re.escape(value)] + elif kind == "cls": + expanded = ["[" + re.escape(value) + "]"] + elif kind == "range": + lo, hi = value.split("\x00") + expanded = ["[" + re.escape(lo) + "-" + re.escape(hi) + "]"] + elif kind == "space": + expanded = [SPACE] + else: + expanded = [{"boundary": _BOUNDARY, "nwb": r"\B", "anchor": value}[kind]] + branches = [b + x for b in branches for x in expanded] + return branches + + +def expr_lookbehind(expr: str) -> str: + raw = _lb_branches(expr) + if any(b == "" for b in raw): + return "" # a zero-width branch always matches: no assertion + if len(raw) == 1: + return f"(?<={raw[0]})" + return "(?:" + "|".join(f"(?<={b})" for b in raw) + ")" + + def expr_neg_lookbehind(expr: str) -> str: - """Negative lookbehind over an expression. Python re requires - fixed-width lookbehinds (Ruby's Onigmo does not), so a top-level - alternation is distributed: (? str: @@ -175,7 +237,7 @@ def expr_to_regex(expr: str) -> str: alts = value.split("\x00") parts.append("(?:" + "|".join(expr_to_regex(a) for a in alts) + ")") elif kind == "opt": - parts.append("(?:" + re.escape(value) + ")?") + parts.append("(?:" + expr_to_regex(value) + ")?") elif kind == "space": parts.append(SPACE) elif kind == "boundary": @@ -200,6 +262,8 @@ def expr_to_literal(expr: str) -> str: parts.append(expr_to_literal(value.split("\x00")[0])) elif kind == "space": parts.append(" ") + elif kind == "opt": + parts.append(expr_to_literal(value)) elif kind in ("boundary", "anchor"): raise ValueError(f"{kind} is not valid in a result expression") return "".join(parts) @@ -222,7 +286,7 @@ def expr_max_length(expr: str) -> int: elif kind == "alt": total += max(expr_max_length(a) for a in value.split("\x00")) elif kind == "opt": - total += len(value) + total += expr_max_length(value) elif kind == "grp": total += expr_max_length(value) # cat: concatenation marker diff --git a/src/interscript/isc.py b/src/interscript/isc.py index 05137f0..68f07ed 100644 --- a/src/interscript/isc.py +++ b/src/interscript/isc.py @@ -22,7 +22,7 @@ import re from pathlib import Path -from .expr import expr_to_regex +from .expr import expr_lookbehind, expr_neg_lookbehind, expr_to_regex class IscParseError(ValueError): @@ -670,10 +670,7 @@ def _render_item(item: dict, aliases: dict[str, str]) -> str: if kind == "capture_group": return "capture(" + _render_item(item["inner"], aliases) + ")" if kind == "maybe": - inner = item["inner"] - if inner["type"] != "string": - raise UnsupportedConstruct("maybe() over non-string items") - return f'maybe("{_escape(inner["value"])}")' + return "maybe(" + _render_item(item["inner"], aliases) + ")" if kind == "primitive": name = item["name"] if name == "space": @@ -794,21 +791,23 @@ def _repl_of(item: dict, aliases: dict[str, str]) -> str: raise UnsupportedConstruct(f"item kind {kind} in a result") -def _guarded_regex(rule: dict, aliases: dict[str, str]) -> str: +def _guarded_regex(rule: dict, aliases: dict[str, str], expr_aliases: dict[str, str]) -> str: """Rule pattern wrapped with its before/after constraints as - lookarounds (the subst-family rendering of parallel constraints).""" + lookarounds (the subst-family rendering of parallel constraints). + Guards render through the expression layer so mixed-width + lookbehinds distribute over branches (Python re is stricter than + Onigmo here).""" prefix = "" suffix = "" for constraint in rule["constraints"]: - fragment = _regex_of(constraint["item"], aliases) if constraint["kind"] == "before": - prefix += f"(?<={fragment})" + prefix += expr_lookbehind(_render_item(constraint["item"], expr_aliases)) elif constraint["kind"] == "not_before": - prefix += f"(? dict: - pattern = _guarded_regex(rule, regex_aliases) + pattern = _guarded_regex(rule, regex_aliases, aliases) to = rule["to"] if to.get("type") == "function" and to.get("name") in _FUNCTIONS: # `to upcase` and friends: the replacement is the match diff --git a/tests/test_engine.py b/tests/test_engine.py index 314750e..1aa2c44 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -212,3 +212,59 @@ def test_any_list_alternatives_are_full_expressions(): # no boundary before "ab" -> the alternative declines, bare b fires. assert e.transliterate("xab") == "xaY" assert e.transliterate("cab") == "caY" + + +def test_before_guard_over_alternation_of_anchor_and_literal(): + """odni-kor: before any([line_start, " "]) — a positive lookbehind + over alternatives of DIFFERENT widths (0 and 1). Python re requires + fixed-width lookbehinds; Ruby's Onigmo does not, so the branches are + distributed: (?<=A|B) == (?:(?<=A)|(?<=B)).""" + import tempfile, os + from interscript import add_load_path, transliterate + + d = tempfile.mkdtemp() + open(os.path.join(d, "lb.isc"), "w").write( + "system \"lb\" {\n" + " stage main {\n" + " parallel {\n" + " sub {\n" + " from \"x\"\n" + " to \"X\"\n" + " before any([line_start, \" \"])\n" + " }\n" + " }\n" + " }\n" + "}\n" + ) + add_load_path(d) + assert transliterate("lb", "x") == "X" + assert transliterate("lb", "a x") == "a X" + assert transliterate("lb", "ax") == "ax" + + +def test_maybe_accepts_full_expressions(): + """alalc-aze / odni-pus: from maybe(any("ab")) + "c" — maybe over a + character class, which the renderer rejected as non-string.""" + import tempfile, os + from interscript import add_load_path, transliterate + + d = tempfile.mkdtemp() + open(os.path.join(d, "mb.isc"), "w").write( + "system \"mb\" {\n" + " stage main {\n" + " parallel {\n" + " sub {\n" + " from maybe(any(\"ab\")) + \"c\"\n" + " to \"Q\"\n" + " }\n" + " }\n" + " }\n" + "}\n" + ) + add_load_path(d) + assert transliterate("mb", "c") == "Q" + assert transliterate("mb", "ac") == "Q" + assert transliterate("mb", "bc") == "Q" + # empty maybe + "c" still matches the bare c (Ruby parallel + # semantics: the rule consumes just "c" here). + assert transliterate("mb", "dc") == "dQ"