1717 r'|any\(\s*"(?P<rlo>(?:[^"\\]|\\.)*)"\s*\.\.\s*"(?P<rhi>(?:[^"\\]|\\.)*)"\s*\)'
1818 r'|any\(\s*"(?P<cls>(?:[^"\\]|\\.)*)"\s*\)'
1919 r'|maybe\(\s*"(?P<opt>(?:[^"\\]|\\.)*)"\s*\)'
20- r'|any\(\s*\[(?P<lst>(?:[^\\\[\]]|\\.)*)\]\s*\)'
2120 r"|(?P<space>\bspace\b)|(?P<boundary>\bboundary\b)"
2221 r"|(?P<nwb>\bnon_word_boundary\b)"
2322 r'|capture\(\s*(?P<grp>(?:[^()\\]|\\.|\([^()]*\))*)\s*\)'
2423 r"|(?P<line_end>\bline_end\b)|(?P<line_start>\bline_start\b)"
2524 r"|(?P<cat>\+)"
2625)
27- _LIST_SPLIT = re .compile (r'"((?:[^"\\]|\\.)*)"' )
26+ def _split_list (src : str ) -> list [str ]:
27+ """Alternatives of any([...]) are full expressions — split on
28+ commas outside quotes and brackets (boundary + "ab" keeps its
29+ boundary atom; any([a, b]) nests)."""
30+ parts , buf , quote , depth = [], [], None , 0
31+ for ch in src :
32+ if quote :
33+ buf .append (ch )
34+ if ch == quote :
35+ quote = None
36+ elif ch in "\" '" :
37+ buf .append (ch )
38+ quote = ch
39+ elif ch == "[" :
40+ depth += 1
41+ buf .append (ch )
42+ elif ch == "]" :
43+ depth -= 1
44+ buf .append (ch )
45+ elif ch == "," and depth == 0 :
46+ parts .append ("" .join (buf ))
47+ buf = []
48+ else :
49+ buf .append (ch )
50+ parts .append ("" .join (buf ))
51+ return [p .strip () for p in parts if p .strip ()]
52+
53+
54+ def _read_bracketed (expr : str , pos : int ) -> tuple [str , int ]:
55+ """With pos at the '[' of any([: return the inner content and the
56+ position after the matching ']' (quote- and depth-aware)."""
57+ depth , i , n = 0 , pos , len (expr )
58+ while i < n :
59+ c = expr [i ]
60+ if c == '"' :
61+ i += 1
62+ while i < n :
63+ if expr [i ] == "\\ " :
64+ i += 2
65+ continue
66+ if expr [i ] == '"' :
67+ break
68+ i += 1
69+ elif c == "[" :
70+ depth += 1
71+ elif c == "]" :
72+ depth -= 1
73+ if depth == 0 :
74+ return expr [pos + 1 : i ], i + 1
75+ i += 1
76+ raise ValueError (f"unterminated any([ in { expr !r} " )
2877_UNESC = re .compile (r"\\u([0-9a-fA-F]{4})" )
2978
3079SPACE = re .escape (" " )
@@ -44,14 +93,30 @@ def _unesc(s: str) -> str:
4493 return _UNESC .sub (lambda m : chr (int (m .group (1 ), 16 )), s )
4594
4695
96+ _ANY_LIST = re .compile (r"any\(\s*\[" )
97+
98+
4799def _scan (expr : str , want : str ):
48100 """Tokenize an expression into (kind, value); raises on gaps."""
49101 out : list [tuple [str , str ]] = []
50102 pos = 0
51- for m in _TOKEN .finditer (expr ):
52- gap = expr [pos : m .start ()]
53- if gap .strip ():
54- raise ValueError (f"cannot parse { want } near { gap .strip ()!r} in { expr !r} " )
103+ while pos < len (expr ):
104+ if expr [pos ].isspace ():
105+ pos += 1
106+ continue
107+ if _ANY_LIST .match (expr , pos ):
108+ bracket = expr .find ("[" , pos )
109+ inner , pos = _read_bracketed (expr , bracket )
110+ while pos < len (expr ) and expr [pos ].isspace ():
111+ pos += 1
112+ if pos >= len (expr ) or expr [pos ] != ")" :
113+ raise ValueError (f"expected ')' closing any([ in { expr !r} " )
114+ pos += 1
115+ out .append (("alt" , "\x00 " .join (_split_list (inner ))))
116+ continue
117+ m = _TOKEN .match (expr , pos )
118+ if not m :
119+ raise ValueError (f"cannot parse { want } near { expr [pos : pos + 20 ]!r} in { expr !r} " )
55120 pos = m .end ()
56121 g = m .groupdict ()
57122 if g ["lit" ] is not None :
@@ -62,9 +127,6 @@ def _scan(expr: str, want: str):
62127 out .append (("cls" , _unesc (g ["cls" ])))
63128 elif g ["opt" ] is not None :
64129 out .append (("opt" , _unesc (g ["opt" ])))
65- elif g ["lst" ] is not None :
66- alts = [_unesc (x ) for x in _LIST_SPLIT .findall (g ["lst" ])]
67- out .append (("alt" , "\x00 " .join (alts )))
68130 elif g ["space" ] is not None :
69131 out .append (("space" , " " ))
70132 elif g ["boundary" ] is not None :
@@ -87,6 +149,18 @@ def _scan(expr: str, want: str):
87149 return out
88150
89151
152+ def expr_neg_lookbehind (expr : str ) -> str :
153+ """Negative lookbehind over an expression. Python re requires
154+ fixed-width lookbehinds (Ruby's Onigmo does not), so a top-level
155+ alternation is distributed: (?<!A|B) == (?<!A)(?<!B)."""
156+ toks = _scan (expr , "expression" )
157+ if len (toks ) == 1 and toks [0 ][0 ] == "alt" :
158+ parts = [expr_to_regex (a ) for a in toks [0 ][1 ].split ("\x00 " )]
159+ else :
160+ parts = [expr_to_regex (expr )]
161+ return "" .join (f"(?<!{ p } )" for p in parts )
162+
163+
90164def expr_to_regex (expr : str ) -> str :
91165 parts = []
92166 for kind , value in _scan (expr , "expression" ):
@@ -99,7 +173,7 @@ def expr_to_regex(expr: str) -> str:
99173 parts .append ("[" + re .escape (lo ) + "-" + re .escape (hi ) + "]" )
100174 elif kind == "alt" :
101175 alts = value .split ("\x00 " )
102- parts .append ("(?:" + "|" .join (re . escape (a ) for a in alts ) + ")" )
176+ parts .append ("(?:" + "|" .join (expr_to_regex (a ) for a in alts ) + ")" )
103177 elif kind == "opt" :
104178 parts .append ("(?:" + re .escape (value ) + ")?" )
105179 elif kind == "space" :
@@ -123,7 +197,7 @@ def expr_to_literal(expr: str) -> str:
123197 elif kind == "cls" :
124198 parts .append (value [0 ]) # deterministic: first alternative
125199 elif kind == "alt" :
126- parts .append (value .split ("\x00 " )[0 ])
200+ parts .append (expr_to_literal ( value .split ("\x00 " )[0 ]) )
127201 elif kind == "space" :
128202 parts .append (" " )
129203 elif kind in ("boundary" , "anchor" ):
@@ -146,7 +220,7 @@ def expr_max_length(expr: str) -> int:
146220 elif kind in ("cls" , "range" , "space" , "boundary" , "nwb" , "anchor" ):
147221 total += 1
148222 elif kind == "alt" :
149- total += max (len (a ) for a in value .split ("\x00 " ))
223+ total += max (expr_max_length (a ) for a in value .split ("\x00 " ))
150224 elif kind == "opt" :
151225 total += len (value )
152226 elif kind == "grp" :
0 commit comments