Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 7 additions & 3 deletions src/interscript/engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@
import re
import unicodedata

from .expr import expr_max_length, expr_to_literal, expr_to_regex, is_plain_string
from .expr import expr_max_length, expr_neg_lookbehind, expr_to_literal, expr_to_regex, is_plain_string


class ExecutionError(ValueError):
Expand All @@ -36,6 +36,10 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str]
full = pat
if sub.get("before"):
full = "(?<=" + expr_to_regex(sub["before"]) + ")" + full
if sub.get("not_before"):
full = expr_neg_lookbehind(sub["not_before"]) + full
if sub.get("not_after"):
full = full + "(?!" + expr_to_regex(sub["not_after"]) + ")"
if sub.get("after"):
full = full + "(?=" + expr_to_regex(sub["after"]) + ")"
key = expr_max_length(sub["pattern"])
Expand All @@ -44,7 +48,7 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str]
key += expr_max_length(sub[guard])
key += sub.get("priority", 0)
indexed.append((key, full, f"s{i}"))
if is_plain_string(sub["pattern"]) and not sub.get("before") and not sub.get("after"):
if is_plain_string(sub["pattern"]) and not any(sub.get(g) for g in ("before", "after", "not_before", "not_after")):
src = expr_to_literal(sub["pattern"])
if src.upper() != src:
anchor_results[f"a{i}"] = expr_to_literal(sub["result"])
Expand All @@ -58,7 +62,7 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str]
casing_map: dict[str, str] = {}
upper_dst: dict[str, str] = {}
for sub in subs:
if is_plain_string(sub["pattern"]) and not sub.get("before") and not sub.get("after"):
if is_plain_string(sub["pattern"]) and not any(sub.get(g) for g in ("before", "after", "not_before", "not_after")):
src = expr_to_literal(sub["pattern"])
dst = expr_to_literal(sub["result"])
casing_map[src] = dst
Expand Down
98 changes: 86 additions & 12 deletions src/interscript/expr.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,14 +17,63 @@
r'|any\(\s*"(?P<rlo>(?:[^"\\]|\\.)*)"\s*\.\.\s*"(?P<rhi>(?:[^"\\]|\\.)*)"\s*\)'
r'|any\(\s*"(?P<cls>(?:[^"\\]|\\.)*)"\s*\)'
r'|maybe\(\s*"(?P<opt>(?:[^"\\]|\\.)*)"\s*\)'
r'|any\(\s*\[(?P<lst>(?:[^\\\[\]]|\\.)*)\]\s*\)'
r"|(?P<space>\bspace\b)|(?P<boundary>\bboundary\b)"
r"|(?P<nwb>\bnon_word_boundary\b)"
r'|capture\(\s*(?P<grp>(?:[^()\\]|\\.|\([^()]*\))*)\s*\)'
r"|(?P<line_end>\bline_end\b)|(?P<line_start>\bline_start\b)"
r"|(?P<cat>\+)"
)
_LIST_SPLIT = re.compile(r'"((?:[^"\\]|\\.)*)"')
def _split_list(src: str) -> list[str]:
"""Alternatives of any([...]) are full expressions — split on
commas outside quotes and brackets (boundary + "ab" keeps its
boundary atom; any([a, b]) nests)."""
parts, buf, quote, depth = [], [], None, 0
for ch in src:
if quote:
buf.append(ch)
if ch == quote:
quote = None
elif ch in "\"'":
buf.append(ch)
quote = ch
elif ch == "[":
depth += 1
buf.append(ch)
elif ch == "]":
depth -= 1
buf.append(ch)
elif ch == "," and depth == 0:
parts.append("".join(buf))
buf = []
else:
buf.append(ch)
parts.append("".join(buf))
return [p.strip() for p in parts if p.strip()]


def _read_bracketed(expr: str, pos: int) -> tuple[str, int]:
"""With pos at the '[' of any([: return the inner content and the
position after the matching ']' (quote- and depth-aware)."""
depth, i, n = 0, pos, len(expr)
while i < n:
c = expr[i]
if c == '"':
i += 1
while i < n:
if expr[i] == "\\":
i += 2
continue
if expr[i] == '"':
break
i += 1
elif c == "[":
depth += 1
elif c == "]":
depth -= 1
if depth == 0:
return expr[pos + 1 : i], i + 1
i += 1
raise ValueError(f"unterminated any([ in {expr!r}")
_UNESC = re.compile(r"\\u([0-9a-fA-F]{4})")

SPACE = re.escape(" ")
Expand All @@ -44,14 +93,30 @@ def _unesc(s: str) -> str:
return _UNESC.sub(lambda m: chr(int(m.group(1), 16)), s)


_ANY_LIST = re.compile(r"any\(\s*\[")


def _scan(expr: str, want: str):
"""Tokenize an expression into (kind, value); raises on gaps."""
out: list[tuple[str, str]] = []
pos = 0
for m in _TOKEN.finditer(expr):
gap = expr[pos : m.start()]
if gap.strip():
raise ValueError(f"cannot parse {want} near {gap.strip()!r} in {expr!r}")
while pos < len(expr):
if expr[pos].isspace():
pos += 1
continue
if _ANY_LIST.match(expr, pos):
bracket = expr.find("[", pos)
inner, pos = _read_bracketed(expr, bracket)
while pos < len(expr) and expr[pos].isspace():
pos += 1
if pos >= len(expr) or expr[pos] != ")":
raise ValueError(f"expected ')' closing any([ in {expr!r}")
pos += 1
out.append(("alt", "\x00".join(_split_list(inner))))
continue
m = _TOKEN.match(expr, pos)
if not m:
raise ValueError(f"cannot parse {want} near {expr[pos : pos + 20]!r} in {expr!r}")
pos = m.end()
g = m.groupdict()
if g["lit"] is not None:
Expand All @@ -62,9 +127,6 @@ def _scan(expr: str, want: str):
out.append(("cls", _unesc(g["cls"])))
elif g["opt"] is not None:
out.append(("opt", _unesc(g["opt"])))
elif g["lst"] is not None:
alts = [_unesc(x) for x in _LIST_SPLIT.findall(g["lst"])]
out.append(("alt", "\x00".join(alts)))
elif g["space"] is not None:
out.append(("space", " "))
elif g["boundary"] is not None:
Expand All @@ -87,6 +149,18 @@ def _scan(expr: str, want: str):
return out


def expr_neg_lookbehind(expr: str) -> str:
"""Negative lookbehind over an expression. Python re requires
fixed-width lookbehinds (Ruby's Onigmo does not), so a top-level
alternation is distributed: (?<!A|B) == (?<!A)(?<!B)."""
toks = _scan(expr, "expression")
if len(toks) == 1 and toks[0][0] == "alt":
parts = [expr_to_regex(a) for a in toks[0][1].split("\x00")]
else:
parts = [expr_to_regex(expr)]
return "".join(f"(?<!{p})" for p in parts)


def expr_to_regex(expr: str) -> str:
parts = []
for kind, value in _scan(expr, "expression"):
Expand All @@ -99,7 +173,7 @@ def expr_to_regex(expr: str) -> str:
parts.append("[" + re.escape(lo) + "-" + re.escape(hi) + "]")
elif kind == "alt":
alts = value.split("\x00")
parts.append("(?:" + "|".join(re.escape(a) for a in alts) + ")")
parts.append("(?:" + "|".join(expr_to_regex(a) for a in alts) + ")")
elif kind == "opt":
parts.append("(?:" + re.escape(value) + ")?")
elif kind == "space":
Expand All @@ -123,7 +197,7 @@ def expr_to_literal(expr: str) -> str:
elif kind == "cls":
parts.append(value[0]) # deterministic: first alternative
elif kind == "alt":
parts.append(value.split("\x00")[0])
parts.append(expr_to_literal(value.split("\x00")[0]))
elif kind == "space":
parts.append(" ")
elif kind in ("boundary", "anchor"):
Expand All @@ -146,7 +220,7 @@ def expr_max_length(expr: str) -> int:
elif kind in ("cls", "range", "space", "boundary", "nwb", "anchor"):
total += 1
elif kind == "alt":
total += max(len(a) for a in value.split("\x00"))
total += max(expr_max_length(a) for a in value.split("\x00"))
elif kind == "opt":
total += len(value)
elif kind == "grp":
Expand Down
38 changes: 37 additions & 1 deletion src/interscript/isc.py
Original file line number Diff line number Diff line change
Expand Up @@ -832,8 +832,12 @@ def _stage_tree(
for constraint in rule["constraints"]:
if constraint["kind"] == "before":
sub["before"] = _render_item(constraint["item"], aliases)
elif constraint["kind"] == "not_before":
sub["not_before"] = _render_item(constraint["item"], aliases)
elif constraint["kind"] == "after":
sub["after"] = _render_item(constraint["item"], aliases)
elif constraint["kind"] == "not_after":
sub["not_after"] = _render_item(constraint["item"], aliases)
subs.append(sub)
if capture_rules:
# Capture-bearing rules degrade to ordered substitutions ahead
Expand Down Expand Up @@ -950,11 +954,43 @@ def _library_aliases(name: str, libs_dir: Path) -> dict[str, str]:
value = m.group(2)
if "'" in value and '"' not in value:
value = value.replace("'", '"')
aliases[m.group(1)] = value
# Library aliases may reference aliases defined earlier
# in the same file (var-kor: jamo = any([jamo_leading_cons,
# ...])); substitute them, outside string literals only.
aliases[m.group(1)] = _subst_alias_refs(value, aliases)
_LIB_CACHE[name] = aliases
return aliases


def _subst_alias_refs(value: str, resolved: dict[str, str]) -> str:
out, i, n = [], 0, len(value)
while i < n:
c = value[i]
if c == '"':
j = i + 1
while j < n:
if value[j] == "\\":
j += 2
continue
if value[j] == '"':
j += 1
break
j += 1
out.append(value[i:j])
i = j
elif c.isascii() and (c.isalnum() or c == "_"):
j = i
while j < n and (value[j].isascii() and (value[j].isalnum() or value[j] in "_-")):
j += 1
word = value[i:j]
out.append(resolved.get(word, word))
i = j
else:
out.append(c)
i += 1
return "".join(out)


def isc_to_tree(source: str, filename: str | None = None, on_unsupported: str = "raise") -> dict:
"""Parse .isc source and convert to the engine's tree shape.

Expand Down
60 changes: 60 additions & 0 deletions tests/test_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -152,3 +152,63 @@ def test_boundary_treats_combining_marks_as_word_chars():
assert Engine(tree).transliterate("dئِ") == "d'i"
# ئ at a true word end: the boundary rule fires.
assert Engine(tree).transliterate("dئ") == "d'a"


@pytest.mark.skipif(not MAPS.is_dir(), reason="interscript maps repo not present")
def test_not_guards_are_match_constraints_not_sort_keys_only():
"""Ruby compiles not_before/not_after as real guards (negative
lookbehind/lookahead — interpreter#build_regexp). The alalc-ara
shape: mas'alah keeps the hamza mark only because أ+fatḥa -> 'a'
declines when ة/ل FOLLOWS (not_after is following context)."""
import tempfile, os
from interscript import add_load_path, transliterate

d = tempfile.mkdtemp()
open(os.path.join(d, "ng.isc"), "w").write(
"system \"ng\" {\n"
" stage main {\n"
" parallel {\n"
" sub {\n"
" from \"xy\"\n"
" to \"A\"\n"
" not_after any(\"ab\")\n"
" }\n"
" sub {\n"
" from \"zw\"\n"
" to \"B\"\n"
" not_before any(\"pq\")\n"
" }\n"
" sub \"x\" \"Q\"\n"
" sub \"z\" \"Z\"\n"
" }\n"
" }\n"
"}\n"
)
add_load_path(d)
# not_after: "xy" followed by a/b declines -> bare x rule fires.
assert transliterate("ng", "xya") == "Qya"
assert transliterate("ng", "xyc") == "Ac"
# not_before: "zw" preceded by p/q declines.
assert transliterate("ng", "pzw") == "pZw"
assert transliterate("ng", "czw") == "cB"


def test_any_list_alternatives_are_full_expressions():
"""any([...]) alternatives are full items (Ruby: Any of Items, each
may be a Group). The list tokenizer kept only quoted strings, so
bare atoms like boundary were silently dropped and not_before
guards lost their boundary branch — alalc-ara word-initial آ then
took the medial "’ā" rule instead of "ā"."""
tree = parse_imp(
'stage {\n parallel {\n'
' sub any([boundary + "ab", "q"]), "X"\n'
' sub "b", "Y"\n'
' }\n}\n'
)
e = Engine(tree)
# boundary-guarded alternative fires at word start.
assert e.transliterate("ab") == "X"
assert e.transliterate("q") == "X"
# no boundary before "ab" -> the alternative declines, bare b fires.
assert e.transliterate("xab") == "xaY"
assert e.transliterate("cab") == "caY"
19 changes: 19 additions & 0 deletions tests/test_isc.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,9 @@

from __future__ import annotations

import os
from pathlib import Path

import pytest

from interscript.isc import IscParseError, UnsupportedConstruct, isc_to_tree
Expand Down Expand Up @@ -267,3 +270,19 @@ def test_imported_alias_resolves():
subst = tree["stages"][0]["children"][0]
assert subst["pattern"] == "404"
assert subst["result"] == "500"


@pytest.mark.skipif(
not Path(os.environ.get("INTERSCRIPT_MAPS_PATH", Path(__file__).parent.parent.parent / "maps" / "maps")).is_dir(),
reason="interscript maps repo not present",
)
def test_library_aliases_resolve_intra_library_references():
"""var-kor defines jamo in terms of other library aliases; the
harvested value must have those substituted (they were previously
dropped silently by the any-list tokenizer)."""
from interscript.isc import _library_aliases

libs = Path(os.environ.get("INTERSCRIPT_MAPS_PATH", Path(__file__).parent.parent.parent / "maps" / "maps")).parent / "libs"
al = _library_aliases("var-kor", libs)
assert "jamo_leading_cons" not in al["jamo"]
assert "\\u1100" in al["jamo"] or "ᄀ" in al["jamo"]
Loading