diff --git a/src/interscript/expr.py b/src/interscript/expr.py index fdfd689..18cd861 100644 --- a/src/interscript/expr.py +++ b/src/interscript/expr.py @@ -17,6 +17,7 @@ r'|any\(\s*"(?P(?:[^"\\]|\\.)*)"\s*\.\.\s*"(?P(?:[^"\\]|\\.)*)"\s*\)' r'|any\(\s*"(?P(?:[^"\\]|\\.)*)"\s*\)' r"|(?P\bspace\b)|(?P\bboundary\b)" + r"|(?P\b(?:alpha|digit|word|any_character)\b)" r"|(?P\bnon_word_boundary\b)" r'|capture\(\s*(?P(?:[^()\\]|\\.|\([^()]*\))*)\s*\)' r"|(?P\bline_end\b)|(?P\bline_start\b)" @@ -77,6 +78,15 @@ def _read_bracketed(expr: str, pos: int) -> tuple[str, int]: SPACE = re.escape(" ") +# Ruby Stdlib::ALIASES, ASCII-exact as Onigmo defines them (Python's +# \w and \d are unicode-wide; Onigmo's are not). +_STDLIB_REGEX = { + "alpha": "[a-zA-Z]", + "digit": "[0-9]", + "word": "[a-zA-Z0-9_]", + "any_character": ".", +} + # Ruby's \b counts combining marks as word characters; Python's \w # does not (Mn is not alphanumeric). At a hamza-carrier + kasra # junction Ruby sees no boundary while Python does — word-final rules @@ -163,6 +173,8 @@ def _scan(expr: str, want: str): out.append(("nwb", "")) elif g["grp"] is not None: out.append(("grp", g["grp"])) + elif g["stdin_kw"] is not None: + out.append(("stdlib", g["stdin_kw"])) elif g["line_end"] is not None: out.append(("anchor", "$")) elif g["line_start"] is not None: @@ -196,6 +208,8 @@ def _lb_branches(expr: str) -> list[str]: expanded = [re.escape(value)] elif kind == "cls": expanded = ["[" + re.escape(value) + "]"] + elif kind == "stdlib": + expanded = [_STDLIB_REGEX[value]] elif kind == "range": lo, hi = value.split("\x00") expanded = ["[" + re.escape(lo) + "-" + re.escape(hi) + "]"] @@ -240,6 +254,8 @@ def expr_to_regex(expr: str) -> str: parts.append("(?:" + expr_to_regex(value) + ")?") elif kind == "space": parts.append(SPACE) + elif kind == "stdlib": + parts.append(_STDLIB_REGEX[value]) elif kind == "boundary": parts.append(_BOUNDARY) elif kind == "nwb": @@ -281,7 +297,7 @@ def expr_max_length(expr: str) -> int: for kind, value in _scan(expr, "expression"): if kind == "lit": total += len(value) - elif kind in ("cls", "range", "space", "boundary", "nwb", "anchor"): + elif kind in ("cls", "range", "space", "boundary", "nwb", "anchor", "stdlib"): total += 1 elif kind == "alt": total += max(expr_max_length(a) for a in value.split("\x00")) diff --git a/src/interscript/isc.py b/src/interscript/isc.py index 68f07ed..21e67bf 100644 --- a/src/interscript/isc.py +++ b/src/interscript/isc.py @@ -53,6 +53,9 @@ def __init__(self, subs: list[dict], capture_rules: list[dict]) -> None: PRIMITIVES = {"boundary", "line_start", "line_end", "word_boundary", "non_word_boundary", "space"} _FUNCTIONS = {"upcase", "downcase", "title_case", "reverse", "strip", "swapcase"} _CONSTRAINTS = {"before", "after", "not_before", "not_after"} +# Stdlib aliases usable as bare names (Ruby Stdlib::ALIASES, resolved +# before doc-local aliases, mirroring the interpreter's lookup order). +_STDLIB_EXPR = {"alpha", "digit", "word", "any_character"} # Tokens that terminate an item inside a rule; a bare word equal to one # of these is a keyword, never an alias reference. _KEYWORDS = _CONSTRAINTS | {"to", "from", "note"} @@ -688,6 +691,8 @@ def _render_item(item: dict, aliases: dict[str, str]) -> str: name = item["name"] if item.get("map"): return _qualified_expr(item["map"], name) + if name in _STDLIB_EXPR: + return name if name in _imported: return _qualified_expr(None, name) if name not in aliases: @@ -750,6 +755,8 @@ def _regex_of(item: dict, aliases: dict[str, str]) -> str: name = item["name"] if item.get("map"): return _qualified_regex(item["map"], name) + if name in _STDLIB_EXPR: + return {"alpha": "[a-zA-Z]", "digit": "[0-9]", "word": "[a-zA-Z0-9_]", "any_character": "."}[name] if name in _imported: return _qualified_regex(None, name) if name not in aliases: @@ -788,6 +795,10 @@ def _repl_of(item: dict, aliases: dict[str, str]) -> str: if name not in aliases: raise UnsupportedConstruct(f"unresolved alias {name}") raise UnsupportedConstruct("alias in a result") + if kind == "primitive": + if item["name"] == "space": + return " " + raise UnsupportedConstruct(f"primitive {item['name']} in a result") raise UnsupportedConstruct(f"item kind {kind} in a result") diff --git a/tests/test_engine.py b/tests/test_engine.py index 6fee48b..9ce9f01 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -281,3 +281,13 @@ def test_maybe_accepts_full_expressions(): # empty maybe + "c" still matches the bare c (Ruby parallel # semantics: the rule consumes just "c" here). assert transliterate("mb", "dc") == "dQ" + + +@pytest.mark.skipif(not MAPS.is_dir(), reason="interscript maps repo not present") +def test_primitive_space_result_pads_the_string(): + """moct-kor pads with `sub line_start space` / `sub line_end space`; + the subst renderer rejected primitive results and silently dropped + the rules, so the before-space guards on initial consonants never + fired: 불국사 -> ᄇulguksa instead of Bulguksa.""" + e = _load("moct-kor-Hang-Latn-2000") + assert e.transliterate("불국사") == "Bulguksa"