Skip to content

Commit d463621

Browse files
authored
Merge pull request #12 from interscript/fix/not-guards-and-list-alternatives
fix: compile not_before/not_after as guards; full expressions in any lists
2 parents 988f240 + 43faa58 commit d463621

5 files changed

Lines changed: 209 additions & 16 deletions

File tree

‎src/interscript/engine.py‎

Lines changed: 7 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -17,7 +17,7 @@
1717
import re
1818
import unicodedata
1919

20-
from .expr import expr_max_length, expr_to_literal, expr_to_regex, is_plain_string
20+
from .expr import expr_max_length, expr_neg_lookbehind, expr_to_literal, expr_to_regex, is_plain_string
2121

2222

2323
class ExecutionError(ValueError):
@@ -36,6 +36,10 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str]
3636
full = pat
3737
if sub.get("before"):
3838
full = "(?<=" + expr_to_regex(sub["before"]) + ")" + full
39+
if sub.get("not_before"):
40+
full = expr_neg_lookbehind(sub["not_before"]) + full
41+
if sub.get("not_after"):
42+
full = full + "(?!" + expr_to_regex(sub["not_after"]) + ")"
3943
if sub.get("after"):
4044
full = full + "(?=" + expr_to_regex(sub["after"]) + ")"
4145
key = expr_max_length(sub["pattern"])
@@ -44,7 +48,7 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str]
4448
key += expr_max_length(sub[guard])
4549
key += sub.get("priority", 0)
4650
indexed.append((key, full, f"s{i}"))
47-
if is_plain_string(sub["pattern"]) and not sub.get("before") and not sub.get("after"):
51+
if is_plain_string(sub["pattern"]) and not any(sub.get(g) for g in ("before", "after", "not_before", "not_after")):
4852
src = expr_to_literal(sub["pattern"])
4953
if src.upper() != src:
5054
anchor_results[f"a{i}"] = expr_to_literal(sub["result"])
@@ -58,7 +62,7 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str]
5862
casing_map: dict[str, str] = {}
5963
upper_dst: dict[str, str] = {}
6064
for sub in subs:
61-
if is_plain_string(sub["pattern"]) and not sub.get("before") and not sub.get("after"):
65+
if is_plain_string(sub["pattern"]) and not any(sub.get(g) for g in ("before", "after", "not_before", "not_after")):
6266
src = expr_to_literal(sub["pattern"])
6367
dst = expr_to_literal(sub["result"])
6468
casing_map[src] = dst

‎src/interscript/expr.py‎

Lines changed: 86 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -17,14 +17,63 @@
1717
r'|any\(\s*"(?P<rlo>(?:[^"\\]|\\.)*)"\s*\.\.\s*"(?P<rhi>(?:[^"\\]|\\.)*)"\s*\)'
1818
r'|any\(\s*"(?P<cls>(?:[^"\\]|\\.)*)"\s*\)'
1919
r'|maybe\(\s*"(?P<opt>(?:[^"\\]|\\.)*)"\s*\)'
20-
r'|any\(\s*\[(?P<lst>(?:[^\\\[\]]|\\.)*)\]\s*\)'
2120
r"|(?P<space>\bspace\b)|(?P<boundary>\bboundary\b)"
2221
r"|(?P<nwb>\bnon_word_boundary\b)"
2322
r'|capture\(\s*(?P<grp>(?:[^()\\]|\\.|\([^()]*\))*)\s*\)'
2423
r"|(?P<line_end>\bline_end\b)|(?P<line_start>\bline_start\b)"
2524
r"|(?P<cat>\+)"
2625
)
27-
_LIST_SPLIT = re.compile(r'"((?:[^"\\]|\\.)*)"')
26+
def _split_list(src: str) -> list[str]:
27+
"""Alternatives of any([...]) are full expressions — split on
28+
commas outside quotes and brackets (boundary + "ab" keeps its
29+
boundary atom; any([a, b]) nests)."""
30+
parts, buf, quote, depth = [], [], None, 0
31+
for ch in src:
32+
if quote:
33+
buf.append(ch)
34+
if ch == quote:
35+
quote = None
36+
elif ch in "\"'":
37+
buf.append(ch)
38+
quote = ch
39+
elif ch == "[":
40+
depth += 1
41+
buf.append(ch)
42+
elif ch == "]":
43+
depth -= 1
44+
buf.append(ch)
45+
elif ch == "," and depth == 0:
46+
parts.append("".join(buf))
47+
buf = []
48+
else:
49+
buf.append(ch)
50+
parts.append("".join(buf))
51+
return [p.strip() for p in parts if p.strip()]
52+
53+
54+
def _read_bracketed(expr: str, pos: int) -> tuple[str, int]:
55+
"""With pos at the '[' of any([: return the inner content and the
56+
position after the matching ']' (quote- and depth-aware)."""
57+
depth, i, n = 0, pos, len(expr)
58+
while i < n:
59+
c = expr[i]
60+
if c == '"':
61+
i += 1
62+
while i < n:
63+
if expr[i] == "\\":
64+
i += 2
65+
continue
66+
if expr[i] == '"':
67+
break
68+
i += 1
69+
elif c == "[":
70+
depth += 1
71+
elif c == "]":
72+
depth -= 1
73+
if depth == 0:
74+
return expr[pos + 1 : i], i + 1
75+
i += 1
76+
raise ValueError(f"unterminated any([ in {expr!r}")
2877
_UNESC = re.compile(r"\\u([0-9a-fA-F]{4})")
2978

3079
SPACE = re.escape(" ")
@@ -44,14 +93,30 @@ def _unesc(s: str) -> str:
4493
return _UNESC.sub(lambda m: chr(int(m.group(1), 16)), s)
4594

4695

96+
_ANY_LIST = re.compile(r"any\(\s*\[")
97+
98+
4799
def _scan(expr: str, want: str):
48100
"""Tokenize an expression into (kind, value); raises on gaps."""
49101
out: list[tuple[str, str]] = []
50102
pos = 0
51-
for m in _TOKEN.finditer(expr):
52-
gap = expr[pos : m.start()]
53-
if gap.strip():
54-
raise ValueError(f"cannot parse {want} near {gap.strip()!r} in {expr!r}")
103+
while pos < len(expr):
104+
if expr[pos].isspace():
105+
pos += 1
106+
continue
107+
if _ANY_LIST.match(expr, pos):
108+
bracket = expr.find("[", pos)
109+
inner, pos = _read_bracketed(expr, bracket)
110+
while pos < len(expr) and expr[pos].isspace():
111+
pos += 1
112+
if pos >= len(expr) or expr[pos] != ")":
113+
raise ValueError(f"expected ')' closing any([ in {expr!r}")
114+
pos += 1
115+
out.append(("alt", "\x00".join(_split_list(inner))))
116+
continue
117+
m = _TOKEN.match(expr, pos)
118+
if not m:
119+
raise ValueError(f"cannot parse {want} near {expr[pos : pos + 20]!r} in {expr!r}")
55120
pos = m.end()
56121
g = m.groupdict()
57122
if g["lit"] is not None:
@@ -62,9 +127,6 @@ def _scan(expr: str, want: str):
62127
out.append(("cls", _unesc(g["cls"])))
63128
elif g["opt"] is not None:
64129
out.append(("opt", _unesc(g["opt"])))
65-
elif g["lst"] is not None:
66-
alts = [_unesc(x) for x in _LIST_SPLIT.findall(g["lst"])]
67-
out.append(("alt", "\x00".join(alts)))
68130
elif g["space"] is not None:
69131
out.append(("space", " "))
70132
elif g["boundary"] is not None:
@@ -87,6 +149,18 @@ def _scan(expr: str, want: str):
87149
return out
88150

89151

152+
def expr_neg_lookbehind(expr: str) -> str:
153+
"""Negative lookbehind over an expression. Python re requires
154+
fixed-width lookbehinds (Ruby's Onigmo does not), so a top-level
155+
alternation is distributed: (?<!A|B) == (?<!A)(?<!B)."""
156+
toks = _scan(expr, "expression")
157+
if len(toks) == 1 and toks[0][0] == "alt":
158+
parts = [expr_to_regex(a) for a in toks[0][1].split("\x00")]
159+
else:
160+
parts = [expr_to_regex(expr)]
161+
return "".join(f"(?<!{p})" for p in parts)
162+
163+
90164
def expr_to_regex(expr: str) -> str:
91165
parts = []
92166
for kind, value in _scan(expr, "expression"):
@@ -99,7 +173,7 @@ def expr_to_regex(expr: str) -> str:
99173
parts.append("[" + re.escape(lo) + "-" + re.escape(hi) + "]")
100174
elif kind == "alt":
101175
alts = value.split("\x00")
102-
parts.append("(?:" + "|".join(re.escape(a) for a in alts) + ")")
176+
parts.append("(?:" + "|".join(expr_to_regex(a) for a in alts) + ")")
103177
elif kind == "opt":
104178
parts.append("(?:" + re.escape(value) + ")?")
105179
elif kind == "space":
@@ -123,7 +197,7 @@ def expr_to_literal(expr: str) -> str:
123197
elif kind == "cls":
124198
parts.append(value[0]) # deterministic: first alternative
125199
elif kind == "alt":
126-
parts.append(value.split("\x00")[0])
200+
parts.append(expr_to_literal(value.split("\x00")[0]))
127201
elif kind == "space":
128202
parts.append(" ")
129203
elif kind in ("boundary", "anchor"):
@@ -146,7 +220,7 @@ def expr_max_length(expr: str) -> int:
146220
elif kind in ("cls", "range", "space", "boundary", "nwb", "anchor"):
147221
total += 1
148222
elif kind == "alt":
149-
total += max(len(a) for a in value.split("\x00"))
223+
total += max(expr_max_length(a) for a in value.split("\x00"))
150224
elif kind == "opt":
151225
total += len(value)
152226
elif kind == "grp":

‎src/interscript/isc.py‎

Lines changed: 37 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -832,8 +832,12 @@ def _stage_tree(
832832
for constraint in rule["constraints"]:
833833
if constraint["kind"] == "before":
834834
sub["before"] = _render_item(constraint["item"], aliases)
835+
elif constraint["kind"] == "not_before":
836+
sub["not_before"] = _render_item(constraint["item"], aliases)
835837
elif constraint["kind"] == "after":
836838
sub["after"] = _render_item(constraint["item"], aliases)
839+
elif constraint["kind"] == "not_after":
840+
sub["not_after"] = _render_item(constraint["item"], aliases)
837841
subs.append(sub)
838842
if capture_rules:
839843
# Capture-bearing rules degrade to ordered substitutions ahead
@@ -950,11 +954,43 @@ def _library_aliases(name: str, libs_dir: Path) -> dict[str, str]:
950954
value = m.group(2)
951955
if "'" in value and '"' not in value:
952956
value = value.replace("'", '"')
953-
aliases[m.group(1)] = value
957+
# Library aliases may reference aliases defined earlier
958+
# in the same file (var-kor: jamo = any([jamo_leading_cons,
959+
# ...])); substitute them, outside string literals only.
960+
aliases[m.group(1)] = _subst_alias_refs(value, aliases)
954961
_LIB_CACHE[name] = aliases
955962
return aliases
956963

957964

965+
def _subst_alias_refs(value: str, resolved: dict[str, str]) -> str:
966+
out, i, n = [], 0, len(value)
967+
while i < n:
968+
c = value[i]
969+
if c == '"':
970+
j = i + 1
971+
while j < n:
972+
if value[j] == "\\":
973+
j += 2
974+
continue
975+
if value[j] == '"':
976+
j += 1
977+
break
978+
j += 1
979+
out.append(value[i:j])
980+
i = j
981+
elif c.isascii() and (c.isalnum() or c == "_"):
982+
j = i
983+
while j < n and (value[j].isascii() and (value[j].isalnum() or value[j] in "_-")):
984+
j += 1
985+
word = value[i:j]
986+
out.append(resolved.get(word, word))
987+
i = j
988+
else:
989+
out.append(c)
990+
i += 1
991+
return "".join(out)
992+
993+
958994
def isc_to_tree(source: str, filename: str | None = None, on_unsupported: str = "raise") -> dict:
959995
"""Parse .isc source and convert to the engine's tree shape.
960996

‎tests/test_engine.py‎

Lines changed: 60 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -152,3 +152,63 @@ def test_boundary_treats_combining_marks_as_word_chars():
152152
assert Engine(tree).transliterate("dئِ") == "d'i"
153153
# ئ at a true word end: the boundary rule fires.
154154
assert Engine(tree).transliterate("dئ") == "d'a"
155+
156+
157+
@pytest.mark.skipif(not MAPS.is_dir(), reason="interscript maps repo not present")
158+
def test_not_guards_are_match_constraints_not_sort_keys_only():
159+
"""Ruby compiles not_before/not_after as real guards (negative
160+
lookbehind/lookahead — interpreter#build_regexp). The alalc-ara
161+
shape: mas'alah keeps the hamza mark only because أ+fatḥa -> 'a'
162+
declines when ة/ل FOLLOWS (not_after is following context)."""
163+
import tempfile, os
164+
from interscript import add_load_path, transliterate
165+
166+
d = tempfile.mkdtemp()
167+
open(os.path.join(d, "ng.isc"), "w").write(
168+
"system \"ng\" {\n"
169+
" stage main {\n"
170+
" parallel {\n"
171+
" sub {\n"
172+
" from \"xy\"\n"
173+
" to \"A\"\n"
174+
" not_after any(\"ab\")\n"
175+
" }\n"
176+
" sub {\n"
177+
" from \"zw\"\n"
178+
" to \"B\"\n"
179+
" not_before any(\"pq\")\n"
180+
" }\n"
181+
" sub \"x\" \"Q\"\n"
182+
" sub \"z\" \"Z\"\n"
183+
" }\n"
184+
" }\n"
185+
"}\n"
186+
)
187+
add_load_path(d)
188+
# not_after: "xy" followed by a/b declines -> bare x rule fires.
189+
assert transliterate("ng", "xya") == "Qya"
190+
assert transliterate("ng", "xyc") == "Ac"
191+
# not_before: "zw" preceded by p/q declines.
192+
assert transliterate("ng", "pzw") == "pZw"
193+
assert transliterate("ng", "czw") == "cB"
194+
195+
196+
def test_any_list_alternatives_are_full_expressions():
197+
"""any([...]) alternatives are full items (Ruby: Any of Items, each
198+
may be a Group). The list tokenizer kept only quoted strings, so
199+
bare atoms like boundary were silently dropped and not_before
200+
guards lost their boundary branch — alalc-ara word-initial آ then
201+
took the medial "’ā" rule instead of "ā"."""
202+
tree = parse_imp(
203+
'stage {\n parallel {\n'
204+
' sub any([boundary + "ab", "q"]), "X"\n'
205+
' sub "b", "Y"\n'
206+
' }\n}\n'
207+
)
208+
e = Engine(tree)
209+
# boundary-guarded alternative fires at word start.
210+
assert e.transliterate("ab") == "X"
211+
assert e.transliterate("q") == "X"
212+
# no boundary before "ab" -> the alternative declines, bare b fires.
213+
assert e.transliterate("xab") == "xaY"
214+
assert e.transliterate("cab") == "caY"

‎tests/test_isc.py‎

Lines changed: 19 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -2,6 +2,9 @@
22

33
from __future__ import annotations
44

5+
import os
6+
from pathlib import Path
7+
58
import pytest
69

710
from interscript.isc import IscParseError, UnsupportedConstruct, isc_to_tree
@@ -267,3 +270,19 @@ def test_imported_alias_resolves():
267270
subst = tree["stages"][0]["children"][0]
268271
assert subst["pattern"] == "404"
269272
assert subst["result"] == "500"
273+
274+
275+
@pytest.mark.skipif(
276+
not Path(os.environ.get("INTERSCRIPT_MAPS_PATH", Path(__file__).parent.parent.parent / "maps" / "maps")).is_dir(),
277+
reason="interscript maps repo not present",
278+
)
279+
def test_library_aliases_resolve_intra_library_references():
280+
"""var-kor defines jamo in terms of other library aliases; the
281+
harvested value must have those substituted (they were previously
282+
dropped silently by the any-list tokenizer)."""
283+
from interscript.isc import _library_aliases
284+
285+
libs = Path(os.environ.get("INTERSCRIPT_MAPS_PATH", Path(__file__).parent.parent.parent / "maps" / "maps")).parent / "libs"
286+
al = _library_aliases("var-kor", libs)
287+
assert "jamo_leading_cons" not in al["jamo"]
288+
assert "\\u1100" in al["jamo"] or "ᄀ" in al["jamo"]

0 commit comments

Comments
 (0)