Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
66 changes: 6 additions & 60 deletions src/interscript/engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,20 +17,18 @@
import re
import unicodedata

from .expr import expr_lookbehind, expr_max_length, expr_neg_lookbehind, expr_to_literal, expr_to_regex, is_plain_string
from .expr import expr_lookbehind, expr_max_length, expr_neg_lookbehind, expr_to_literal, expr_to_regex


class ExecutionError(ValueError):
"""The map uses a construct this engine does not implement yet."""


def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str], dict[str, str]]:
def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str]]:
"""Compile one parallel group: longest-pattern-first alternation
with a named group per sub; lookaround guards for before:/after:.
Plain-string patterns additionally feed the casing maps."""
indexed = []
anchor_results: dict[str, str] = {}
n = len(subs)
for i, sub in enumerate(subs):
pat = expr_to_regex(sub["pattern"])
full = pat
Expand All @@ -48,27 +46,12 @@ def _compile_parallel(subs: list[dict]) -> tuple[re.Pattern[str], dict[str, str]
key += expr_max_length(sub[guard])
key += sub.get("priority", 0)
indexed.append((key, full, f"s{i}"))
if is_plain_string(sub["pattern"]) and not any(sub.get(g) for g in ("before", "after", "not_before", "not_after")):
src = expr_to_literal(sub["pattern"])
if src.upper() != src:
anchor_results[f"a{i}"] = expr_to_literal(sub["result"])
indexed.append((len(src), re.escape(src.upper()), f"a{i}"))
indexed.sort(key=lambda t: -t[0])
combined = "|".join(f"(?P<{name}>{full})" for _, full, name in indexed)
pattern = re.compile(combined) if indexed else re.compile(r"(?!)")

results = {f"s{i}": expr_to_literal(sub["result"]) for i, sub in enumerate(subs)}
results.update(anchor_results)
casing_map: dict[str, str] = {}
upper_dst: dict[str, str] = {}
for sub in subs:
if is_plain_string(sub["pattern"]) and not any(sub.get(g) for g in ("before", "after", "not_before", "not_after")):
src = expr_to_literal(sub["pattern"])
dst = expr_to_literal(sub["result"])
casing_map[src] = dst
if src.upper() != src:
upper_dst[src.upper()] = dst
return pattern, {"casing": casing_map, "upper": upper_dst, "results": results}, {}
return pattern, results


_CASE_FNS = {
Expand All @@ -89,7 +72,6 @@ def __init__(self, tree: dict, loader=None, on_unsupported: str = "raise") -> No
self.on_unsupported = on_unsupported
self.skipped_unsupported: list[str] = []
self._compiled: re.Pattern[str] | None = None
self._compiled_map: dict[str, str] = {}
self._group_results: dict[str, str] = {}
self._compiled_source: int | None = None

Expand All @@ -103,51 +85,15 @@ def _run_stage(self, stage: dict, text: str) -> str:
text = self._run_op(child, text)
return text

def _group_repl(self, m: re.Match[str], text: str) -> str:
name = m.lastgroup if m.lastgroup else ""
if name in self._group_results:
result = self._group_results[name]
tok = m.group(0)
if result != result.upper() and tok == tok.upper() and tok != tok.lower():
ws, we = m.start(), m.end()
while ws > 0 and text[ws - 1].isalpha():
ws -= 1
while we < len(text) and text[we].isalpha():
we += 1
if text[ws:we].isupper():
return result.upper()
return result
return self._parallel_repl(m, text)

def _parallel_repl(self, m: re.Match[str], text: str) -> str:
tok = m.group(0)
dst = self._compiled_map.get(tok) or self._upper_dst.get(tok)
if dst is None:
return tok
# interscript-ruby casing convention: inside an ALL-CAPS source
# word, a fully-uppercase source token uppercases its result
# (Я -> Ya normally, YA inside БЯГА).
if dst != dst.upper() and tok == tok.upper() and tok != tok.lower():
ws, we = m.start(), m.end()
while ws > 0 and text[ws - 1].isalpha():
ws -= 1
while we < len(text) and text[we].isalpha():
we += 1
if text[ws:we].isupper():
return dst.upper()
return dst

def _run_op(self, op: dict, text: str) -> str:
kind = op.get("kind")
if kind == "parallel":
if self._compiled is None or self._compiled_source != id(op):
pattern, maps, _ = _compile_parallel(op["subs"])
pattern, results = _compile_parallel(op["subs"])
self._compiled = pattern
self._compiled_map = maps["casing"]
self._upper_dst = maps["upper"]
self._group_results = maps["results"]
self._group_results = results
self._compiled_source = id(op)
return self._compiled.sub(lambda m: self._group_repl(m, text), text)
return self._compiled.sub(lambda m: self._group_results[m.lastgroup], text)
if kind == "subst":
flags = re.IGNORECASE if op.get("ignore_case") else 0
pattern = re.compile(op["pattern"], flags)
Expand Down
17 changes: 15 additions & 2 deletions tests/test_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -47,15 +47,28 @@ def test_parallel_substitution():
assert engine.transliterate("say hello") == "say HELLO"


def test_all_caps_word_uppercases_result():
def test_unmapped_uppercase_passes_through_like_ruby():
"""Measured against the Ruby interpreter: there is NO implicit
casing convention. With rules for Б Г А б г а but none for Я,
БЯГА -> BЯGA (Я passes through), Бяга -> Byaga."""
tree2 = parse_imp(
'stage {\n parallel {\n sub "Б", "B"\n sub "я", "ya"\n'
' sub "Г", "G"\n sub "А", "A"\n sub "г", "g"\n sub "а", "a"\n }\n}\n'
)
assert Engine(tree2).transliterate("БЯГА") == "BYAGA"
assert Engine(tree2).transliterate("БЯГА") == "BЯGA"
assert Engine(tree2).transliterate("Бяга") == "Byaga"


@pytest.mark.skipif(not MAPS.is_dir(), reason="interscript maps repo not present")
def test_explicit_uppercase_rules_not_shadowed_by_implicit_casing():
"""bgnpcgn-ukr carries explicit sub "А" "A" rules; the engine's
implicit casing anchors for lowercase rules shadowed them, so
Авдіївська came out lowercase."""
e = _load("bgnpcgn-ukr-Cyrl-Latn-1965")
assert e.transliterate("Авдіївська") == "Avdiyivs’ka"
assert e.transliterate("Міськрада") == "Mis’krada"


def test_unsupported_construct_raises_by_default():
tree = parse_imp('stage {\n secryst "model"\n}\n')
with pytest.raises(ExecutionError):
Expand Down
Loading