Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions src/interscript/engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@
import re
import unicodedata

from .expr import expr_max_length, expr_neg_lookbehind, expr_to_literal, expr_to_regex, is_plain_string
from .expr import expr_lookbehind, expr_max_length, expr_neg_lookbehind, expr_to_literal, expr_to_regex, is_plain_string


class ExecutionError(ValueError):
Expand All @@ -35,7 +35,7 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str]
pat = expr_to_regex(sub["pattern"])
full = pat
if sub.get("before"):
full = "(?<=" + expr_to_regex(sub["before"]) + ")" + full
full = expr_lookbehind(sub["before"]) + full
if sub.get("not_before"):
full = expr_neg_lookbehind(sub["not_before"]) + full
if sub.get("not_after"):
Expand Down
92 changes: 78 additions & 14 deletions src/interscript/expr.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,6 @@
r'"(?P<lit>(?:[^"\\]|\\.)*)"'
r'|any\(\s*"(?P<rlo>(?:[^"\\]|\\.)*)"\s*\.\.\s*"(?P<rhi>(?:[^"\\]|\\.)*)"\s*\)'
r'|any\(\s*"(?P<cls>(?:[^"\\]|\\.)*)"\s*\)'
r'|maybe\(\s*"(?P<opt>(?:[^"\\]|\\.)*)"\s*\)'
r"|(?P<space>\bspace\b)|(?P<boundary>\bboundary\b)"
r"|(?P<nwb>\bnon_word_boundary\b)"
r'|capture\(\s*(?P<grp>(?:[^()\\]|\\.|\([^()]*\))*)\s*\)'
Expand Down Expand Up @@ -94,6 +93,32 @@ def _unesc(s: str) -> str:


_ANY_LIST = re.compile(r"any\(\s*\[")
_MAYBE = re.compile(r"maybe\(\s*")


def _read_parenthesized(expr: str, pos: int) -> tuple[str, int]:
"""With pos at '(': return the inner content and the position after
the matching ')' (quote-, bracket- and depth-aware)."""
depth, i, n = 0, pos, len(expr)
while i < n:
c = expr[i]
if c == '"':
i += 1
while i < n:
if expr[i] == "\\":
i += 2
continue
if expr[i] == '"':
break
i += 1
elif c == "(":
depth += 1
elif c == ")":
depth -= 1
if depth == 0:
return expr[pos + 1 : i], i + 1
i += 1
raise ValueError(f"unterminated parenthesis in {expr!r}")


def _scan(expr: str, want: str):
Expand All @@ -104,6 +129,11 @@ def _scan(expr: str, want: str):
if expr[pos].isspace():
pos += 1
continue
if _MAYBE.match(expr, pos):
paren = expr.find("(", pos)
inner, pos = _read_parenthesized(expr, paren)
out.append(("opt", inner))
continue
if _ANY_LIST.match(expr, pos):
bracket = expr.find("[", pos)
inner, pos = _read_bracketed(expr, bracket)
Expand All @@ -125,8 +155,6 @@ def _scan(expr: str, want: str):
out.append(("range", _unesc(g["rlo"]) + "\x00" + _unesc(g["rhi"])))
elif g["cls"] is not None:
out.append(("cls", _unesc(g["cls"])))
elif g["opt"] is not None:
out.append(("opt", _unesc(g["opt"])))
elif g["space"] is not None:
out.append(("space", " "))
elif g["boundary"] is not None:
Expand All @@ -149,16 +177,50 @@ def _scan(expr: str, want: str):
return out


def _lb_branches(expr: str) -> list[str]:
"""Expand an expression into fixed-width regex branches by cross
product over alternations and optionals. Python re requires
fixed-width lookbehinds where Ruby's Onigmo does not, so guards
distribute over their branches instead."""
branches = [""]
for kind, value in _scan(expr, "lookbehind"):
if kind == "cat":
continue
if kind == "alt":
expanded = [b for a in value.split("\x00") for b in _lb_branches(a)]
elif kind == "opt":
expanded = _lb_branches(value) + [""]
elif kind == "grp":
expanded = ["(" + expr_to_regex(value) + ")"]
elif kind == "lit":
expanded = [re.escape(value)]
elif kind == "cls":
expanded = ["[" + re.escape(value) + "]"]
elif kind == "range":
lo, hi = value.split("\x00")
expanded = ["[" + re.escape(lo) + "-" + re.escape(hi) + "]"]
elif kind == "space":
expanded = [SPACE]
else:
expanded = [{"boundary": _BOUNDARY, "nwb": r"\B", "anchor": value}[kind]]
branches = [b + x for b in branches for x in expanded]
return branches


def expr_lookbehind(expr: str) -> str:
raw = _lb_branches(expr)
if any(b == "" for b in raw):
return "" # a zero-width branch always matches: no assertion
if len(raw) == 1:
return f"(?<={raw[0]})"
return "(?:" + "|".join(f"(?<={b})" for b in raw) + ")"


def expr_neg_lookbehind(expr: str) -> str:
"""Negative lookbehind over an expression. Python re requires
fixed-width lookbehinds (Ruby's Onigmo does not), so a top-level
alternation is distributed: (?<!A|B) == (?<!A)(?<!B)."""
toks = _scan(expr, "expression")
if len(toks) == 1 and toks[0][0] == "alt":
parts = [expr_to_regex(a) for a in toks[0][1].split("\x00")]
else:
parts = [expr_to_regex(expr)]
return "".join(f"(?<!{p})" for p in parts)
raw = _lb_branches(expr)
if any(b == "" for b in raw):
return "(?!)" # a zero-width branch always matches: never fires
return "".join(f"(?<!{b})" for b in raw)


def expr_to_regex(expr: str) -> str:
Expand All @@ -175,7 +237,7 @@ def expr_to_regex(expr: str) -> str:
alts = value.split("\x00")
parts.append("(?:" + "|".join(expr_to_regex(a) for a in alts) + ")")
elif kind == "opt":
parts.append("(?:" + re.escape(value) + ")?")
parts.append("(?:" + expr_to_regex(value) + ")?")
elif kind == "space":
parts.append(SPACE)
elif kind == "boundary":
Expand All @@ -200,6 +262,8 @@ def expr_to_literal(expr: str) -> str:
parts.append(expr_to_literal(value.split("\x00")[0]))
elif kind == "space":
parts.append(" ")
elif kind == "opt":
parts.append(expr_to_literal(value))
elif kind in ("boundary", "anchor"):
raise ValueError(f"{kind} is not valid in a result expression")
return "".join(parts)
Expand All @@ -222,7 +286,7 @@ def expr_max_length(expr: str) -> int:
elif kind == "alt":
total += max(expr_max_length(a) for a in value.split("\x00"))
elif kind == "opt":
total += len(value)
total += expr_max_length(value)
elif kind == "grp":
total += expr_max_length(value)
# cat: concatenation marker
Expand Down
25 changes: 12 additions & 13 deletions src/interscript/isc.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@
import re
from pathlib import Path

from .expr import expr_to_regex
from .expr import expr_lookbehind, expr_neg_lookbehind, expr_to_regex


class IscParseError(ValueError):
Expand Down Expand Up @@ -670,10 +670,7 @@ def _render_item(item: dict, aliases: dict[str, str]) -> str:
if kind == "capture_group":
return "capture(" + _render_item(item["inner"], aliases) + ")"
if kind == "maybe":
inner = item["inner"]
if inner["type"] != "string":
raise UnsupportedConstruct("maybe() over non-string items")
return f'maybe("{_escape(inner["value"])}")'
return "maybe(" + _render_item(item["inner"], aliases) + ")"
if kind == "primitive":
name = item["name"]
if name == "space":
Expand Down Expand Up @@ -794,21 +791,23 @@ def _repl_of(item: dict, aliases: dict[str, str]) -> str:
raise UnsupportedConstruct(f"item kind {kind} in a result")


def _guarded_regex(rule: dict, aliases: dict[str, str]) -> str:
def _guarded_regex(rule: dict, aliases: dict[str, str], expr_aliases: dict[str, str]) -> str:
"""Rule pattern wrapped with its before/after constraints as
lookarounds (the subst-family rendering of parallel constraints)."""
lookarounds (the subst-family rendering of parallel constraints).
Guards render through the expression layer so mixed-width
lookbehinds distribute over branches (Python re is stricter than
Onigmo here)."""
prefix = ""
suffix = ""
for constraint in rule["constraints"]:
fragment = _regex_of(constraint["item"], aliases)
if constraint["kind"] == "before":
prefix += f"(?<={fragment})"
prefix += expr_lookbehind(_render_item(constraint["item"], expr_aliases))
elif constraint["kind"] == "not_before":
prefix += f"(?<!{fragment})"
prefix += expr_neg_lookbehind(_render_item(constraint["item"], expr_aliases))
elif constraint["kind"] == "after":
suffix += f"(?={fragment})"
suffix += "(?=" + _regex_of(constraint["item"], aliases) + ")"
elif constraint["kind"] == "not_after":
suffix += f"(?!{fragment})"
suffix += "(?!" + _regex_of(constraint["item"], aliases) + ")"
return prefix + _regex_of(rule["from"], aliases) + suffix


Expand Down Expand Up @@ -1042,7 +1041,7 @@ def isc_to_tree(source: str, filename: str | None = None, on_unsupported: str =
_imported.setdefault(name, d["target"])

def _subst(rule: dict) -> dict:
pattern = _guarded_regex(rule, regex_aliases)
pattern = _guarded_regex(rule, regex_aliases, aliases)
to = rule["to"]
if to.get("type") == "function" and to.get("name") in _FUNCTIONS:
# `to upcase` and friends: the replacement is the match
Expand Down
56 changes: 56 additions & 0 deletions tests/test_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -212,3 +212,59 @@ def test_any_list_alternatives_are_full_expressions():
# no boundary before "ab" -> the alternative declines, bare b fires.
assert e.transliterate("xab") == "xaY"
assert e.transliterate("cab") == "caY"


def test_before_guard_over_alternation_of_anchor_and_literal():
"""odni-kor: before any([line_start, " "]) — a positive lookbehind
over alternatives of DIFFERENT widths (0 and 1). Python re requires
fixed-width lookbehinds; Ruby's Onigmo does not, so the branches are
distributed: (?<=A|B) == (?:(?<=A)|(?<=B))."""
import tempfile, os
from interscript import add_load_path, transliterate

d = tempfile.mkdtemp()
open(os.path.join(d, "lb.isc"), "w").write(
"system \"lb\" {\n"
" stage main {\n"
" parallel {\n"
" sub {\n"
" from \"x\"\n"
" to \"X\"\n"
" before any([line_start, \" \"])\n"
" }\n"
" }\n"
" }\n"
"}\n"
)
add_load_path(d)
assert transliterate("lb", "x") == "X"
assert transliterate("lb", "a x") == "a X"
assert transliterate("lb", "ax") == "ax"


def test_maybe_accepts_full_expressions():
"""alalc-aze / odni-pus: from maybe(any("ab")) + "c" — maybe over a
character class, which the renderer rejected as non-string."""
import tempfile, os
from interscript import add_load_path, transliterate

d = tempfile.mkdtemp()
open(os.path.join(d, "mb.isc"), "w").write(
"system \"mb\" {\n"
" stage main {\n"
" parallel {\n"
" sub {\n"
" from maybe(any(\"ab\")) + \"c\"\n"
" to \"Q\"\n"
" }\n"
" }\n"
" }\n"
"}\n"
)
add_load_path(d)
assert transliterate("mb", "c") == "Q"
assert transliterate("mb", "ac") == "Q"
assert transliterate("mb", "bc") == "Q"
# empty maybe + "c" still matches the bare c (Ruby parallel
# semantics: the rule consumes just "c" here).
assert transliterate("mb", "dc") == "dQ"
Loading