Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 17 additions & 1 deletion src/interscript/expr.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@
r'|any\(\s*"(?P<rlo>(?:[^"\\]|\\.)*)"\s*\.\.\s*"(?P<rhi>(?:[^"\\]|\\.)*)"\s*\)'
r'|any\(\s*"(?P<cls>(?:[^"\\]|\\.)*)"\s*\)'
r"|(?P<space>\bspace\b)|(?P<boundary>\bboundary\b)"
r"|(?P<stdin_kw>\b(?:alpha|digit|word|any_character)\b)"
r"|(?P<nwb>\bnon_word_boundary\b)"
r'|capture\(\s*(?P<grp>(?:[^()\\]|\\.|\([^()]*\))*)\s*\)'
r"|(?P<line_end>\bline_end\b)|(?P<line_start>\bline_start\b)"
Expand Down Expand Up @@ -77,6 +78,15 @@ def _read_bracketed(expr: str, pos: int) -> tuple[str, int]:

SPACE = re.escape(" ")

# Ruby Stdlib::ALIASES, ASCII-exact as Onigmo defines them (Python's
# \w and \d are unicode-wide; Onigmo's are not).
_STDLIB_REGEX = {
"alpha": "[a-zA-Z]",
"digit": "[0-9]",
"word": "[a-zA-Z0-9_]",
"any_character": ".",
}

# Ruby's \b counts combining marks as word characters; Python's \w
# does not (Mn is not alphanumeric). At a hamza-carrier + kasra
# junction Ruby sees no boundary while Python does — word-final rules
Expand Down Expand Up @@ -163,6 +173,8 @@ def _scan(expr: str, want: str):
out.append(("nwb", ""))
elif g["grp"] is not None:
out.append(("grp", g["grp"]))
elif g["stdin_kw"] is not None:
out.append(("stdlib", g["stdin_kw"]))
elif g["line_end"] is not None:
out.append(("anchor", "$"))
elif g["line_start"] is not None:
Expand Down Expand Up @@ -196,6 +208,8 @@ def _lb_branches(expr: str) -> list[str]:
expanded = [re.escape(value)]
elif kind == "cls":
expanded = ["[" + re.escape(value) + "]"]
elif kind == "stdlib":
expanded = [_STDLIB_REGEX[value]]
elif kind == "range":
lo, hi = value.split("\x00")
expanded = ["[" + re.escape(lo) + "-" + re.escape(hi) + "]"]
Expand Down Expand Up @@ -240,6 +254,8 @@ def expr_to_regex(expr: str) -> str:
parts.append("(?:" + expr_to_regex(value) + ")?")
elif kind == "space":
parts.append(SPACE)
elif kind == "stdlib":
parts.append(_STDLIB_REGEX[value])
elif kind == "boundary":
parts.append(_BOUNDARY)
elif kind == "nwb":
Expand Down Expand Up @@ -281,7 +297,7 @@ def expr_max_length(expr: str) -> int:
for kind, value in _scan(expr, "expression"):
if kind == "lit":
total += len(value)
elif kind in ("cls", "range", "space", "boundary", "nwb", "anchor"):
elif kind in ("cls", "range", "space", "boundary", "nwb", "anchor", "stdlib"):
total += 1
elif kind == "alt":
total += max(expr_max_length(a) for a in value.split("\x00"))
Expand Down
11 changes: 11 additions & 0 deletions src/interscript/isc.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,9 @@ def __init__(self, subs: list[dict], capture_rules: list[dict]) -> None:
PRIMITIVES = {"boundary", "line_start", "line_end", "word_boundary", "non_word_boundary", "space"}
_FUNCTIONS = {"upcase", "downcase", "title_case", "reverse", "strip", "swapcase"}
_CONSTRAINTS = {"before", "after", "not_before", "not_after"}
# Stdlib aliases usable as bare names (Ruby Stdlib::ALIASES, resolved
# before doc-local aliases, mirroring the interpreter's lookup order).
_STDLIB_EXPR = {"alpha", "digit", "word", "any_character"}
# Tokens that terminate an item inside a rule; a bare word equal to one
# of these is a keyword, never an alias reference.
_KEYWORDS = _CONSTRAINTS | {"to", "from", "note"}
Expand Down Expand Up @@ -688,6 +691,8 @@ def _render_item(item: dict, aliases: dict[str, str]) -> str:
name = item["name"]
if item.get("map"):
return _qualified_expr(item["map"], name)
if name in _STDLIB_EXPR:
return name
if name in _imported:
return _qualified_expr(None, name)
if name not in aliases:
Expand Down Expand Up @@ -750,6 +755,8 @@ def _regex_of(item: dict, aliases: dict[str, str]) -> str:
name = item["name"]
if item.get("map"):
return _qualified_regex(item["map"], name)
if name in _STDLIB_EXPR:
return {"alpha": "[a-zA-Z]", "digit": "[0-9]", "word": "[a-zA-Z0-9_]", "any_character": "."}[name]
if name in _imported:
return _qualified_regex(None, name)
if name not in aliases:
Expand Down Expand Up @@ -788,6 +795,10 @@ def _repl_of(item: dict, aliases: dict[str, str]) -> str:
if name not in aliases:
raise UnsupportedConstruct(f"unresolved alias {name}")
raise UnsupportedConstruct("alias in a result")
if kind == "primitive":
if item["name"] == "space":
return " "
raise UnsupportedConstruct(f"primitive {item['name']} in a result")
raise UnsupportedConstruct(f"item kind {kind} in a result")


Expand Down
10 changes: 10 additions & 0 deletions tests/test_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -281,3 +281,13 @@ def test_maybe_accepts_full_expressions():
# empty maybe + "c" still matches the bare c (Ruby parallel
# semantics: the rule consumes just "c" here).
assert transliterate("mb", "dc") == "dQ"


@pytest.mark.skipif(not MAPS.is_dir(), reason="interscript maps repo not present")
def test_primitive_space_result_pads_the_string():
"""moct-kor pads with `sub line_start space` / `sub line_end space`;
the subst renderer rejected primitive results and silently dropped
the rules, so the before-space guards on initial consonants never
fired: 불국사 -> ᄇulguksa instead of Bulguksa."""
e = _load("moct-kor-Hang-Latn-2000")
assert e.transliterate("불국사") == "Bulguksa"
Loading