Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
29 changes: 20 additions & 9 deletions src/interscript/expr.py
Original file line number Diff line number Diff line change
Expand Up @@ -87,14 +87,15 @@ def _read_bracketed(expr: str, pos: int) -> tuple[str, int]:
"any_character": ".",
}

# Ruby's \b counts combining marks as word characters; Python's \w
# does not (Mn is not alphanumeric). At a hamza-carrier + kasra
# junction Ruby sees no boundary while Python does — word-final rules
# fired wrongly and doubled vowels. Express the boundary as an
# explicit word/non-word transition over a word class that includes
# combining marks (Mnemonic ranges: combining diacritics 0300-036F,
# Arabic diacritics 064B-065F, 0670, and Quranic annotation 06D6-06ED).
_WORD = r"[\w\u0300-\u036F\u064B-\u065F\u0670\u06D6-\u06ED]"
# Ruby's \b is Unicode-aware (the Word property: letters, MARKS,
# digits, connectors) while its \w stays ASCII-only; Python's \b uses
# \w, which excludes combining marks. At a hamza-carrier + kasra or
# क + anusvara junction Ruby therefore sees no boundary where Python
# does, and word-final rules fired wrongly (dā'aim for dā'im,
# kṁganā for kaṁganā). The boundary is expressed explicitly over a
# word class that adds every Mark range (290 ranges, generated) to
# \w, plus the Join_Control characters.
_WORD = "[\\ẁ-ͯ҃-҉֑-ֽֿׁ-ׂׄ-ׇׅؐ-ًؚ-ٰٟۖ-ۜ۟-ۤۧ-۪ۨ-ܑۭܰ-݊ަ-ް߫-߽߳ࠖ-࠙ࠛ-ࠣࠥ-ࠧࠩ-࡙࠭-࡛࣓-ࣣ࣡-ःऺ-़ा-ॏ॑-ॗॢ-ॣঁ-ঃ়া-ৄে-ৈো-্ৗৢ-ৣ৾ਁ-ਃ਼ਾ-ੂੇ-ੈੋ-੍ੑੰ-ੱੵઁ-ઃ઼ા-ૅે-ૉો-્ૢ-ૣૺ-૿ଁ-ଃ଼ା-ୄେ-ୈୋ-୍୕-ୗୢ-ୣஂா-ூெ-ைொ-்ௗఀ-ఄా-ౄె-ైొ-్ౕ-ౖౢ-ౣಁ-ಃ಼ಾ-ೄೆ-ೈೊ-್ೕ-ೖೢ-ೣഀ-ഃ഻-഼ാ-ൄെ-ൈൊ-്ൗൢ-ൣඁ-ඃ්ා-ුූෘ-ෟෲ-ෳัิ-ฺ็-๎ັິ-ຼ່-ໍ༘-༹༙༵༷༾-༿ཱ-྄྆-྇ྍ-ྗྙ-ྼ࿆ါ-ှၖ-ၙၞ-ၠၢ-ၤၧ-ၭၱ-ၴႂ-ႍႏႚ-ႝ፝-፟ᜒ-᜔ᜲ-᜴ᝒ-ᝓᝲ-ᝳ឴-៓៝᠋-᠍ᢅ-ᢆᢩᤠ-ᤫᤰ-᤻ᨗ-ᨛᩕ-ᩞ᩠-᩿᩼᪰-ᫀᬀ-ᬄ᬴-᭄᭫-᭳ᮀ-ᮂᮡ-ᮭ᯦-᯳ᰤ-᰷᳐-᳔᳒-᳨᳭᳴᳷-᳹᷀-᷹᷻-᷿⃐-⃰⳯-⵿⳱ⷠ-〪ⷿ-゙〯-゚꙯-꙲ꙴ-꙽ꚞ-ꚟ꛰-꛱ꠂ꠆ꠋꠣ-ꠧ꠬ꢀ-ꢁꢴ-ꣅ꣠-꣱ꣿꤦ-꤭ꥇ-꥓ꦀ-ꦃ꦳-꧀ꧥꨩ-ꨶꩃꩌ-ꩍꩻ-ꩽꪰꪲ-ꪴꪷ-ꪸꪾ-꪿꫁ꫫ-ꫯꫵ-꫶ꯣ-ꯪ꯬-꯭ﬞ︀-️︠-𐇽𐋠︯𐍶-𐍺𐨁-𐨃𐨅-𐨆𐨌-𐨏𐨸-𐨿𐨺𐫥-𐫦𐴤-𐴧𐺫-𐽆𐺬-𐽐𑀀-𑀂𑀸-𑁆𑁿-𑂂𑂰-𑂺𑄀-𑄂𑄧-𑄴𑅅-𑅆𑅳𑆀-𑆂𑆳-𑇀𑇉-𑇌𑇎-𑇏𑈬-𑈷𑈾𑋟-𑋪𑌀-𑌃𑌻-𑌼𑌾-𑍄𑍇-𑍈𑍋-𑍍𑍗𑍢-𑍣𑍦-𑍬𑍰-𑍴𑐵-𑑆𑑞𑒰-𑓃𑖯-𑖵𑖸-𑗀𑗜-𑗝𑘰-𑙀𑚫-𑚷𑜝-𑜫𑠬-𑠺𑤰-𑤵𑤷-𑤸𑤻-𑤾𑥀𑥂-𑥃𑧑-𑧗𑧚-𑧠𑧤𑨁-𑨊𑨳-𑨹𑨻-𑨾𑩇𑩑-𑩛𑪊-𑪙𑰯-𑰶𑰸-𑰿𑲒-𑲧𑲩-𑲶𑴱-𑴶𑴺𑴼-𑴽𑴿-𑵅𑵇𑶊-𑶎𑶐-𑶑𑶓-𑶗𑻳-𑻶𖫰-𖫴𖬰-𖬶𖽏𖽑-𖾇𖾏-𖾒𖿤𖿰-𖿱𛲝-𛲞𝅥-𝅩𝅭-𝅲𝅻-𝆂𝆅-𝆋𝆪-𝆭𝉂-𝉄𝨀-𝨶𝨻-𝩬𝩵𝪄𝪛-𝪟𝪡-𝪯𞀀-𞀆𞀈-𞀘𞀛-𞀡𞀣-𞀤𞀦-𞀪𞄰-𞄶𞋬-𞣐𞋯-𞣖𞥄-𞥊󠄀-󠇯\u200C\u200D]"
_BOUNDARY = "(?:(?<=" + _WORD + ")(?!" + _WORD + ")|(?<!" + _WORD + ")(?=" + _WORD + "))"


Expand All @@ -104,6 +105,7 @@ def _unesc(s: str) -> str:

_ANY_LIST = re.compile(r"any\(\s*\[")
_MAYBE = re.compile(r"maybe\(\s*")
_SOME = re.compile(r"some\(\s*")


def _read_parenthesized(expr: str, pos: int) -> tuple[str, int]:
Expand Down Expand Up @@ -139,6 +141,11 @@ def _scan(expr: str, want: str):
if expr[pos].isspace():
pos += 1
continue
if _SOME.match(expr, pos):
paren = expr.find("(", pos)
inner, pos = _read_parenthesized(expr, paren)
out.append(("rep", inner))
continue
if _MAYBE.match(expr, pos):
paren = expr.find("(", pos)
inner, pos = _read_parenthesized(expr, paren)
Expand Down Expand Up @@ -252,6 +259,8 @@ def expr_to_regex(expr: str) -> str:
parts.append("(?:" + "|".join(expr_to_regex(a) for a in alts) + ")")
elif kind == "opt":
parts.append("(?:" + expr_to_regex(value) + ")?")
elif kind == "rep":
parts.append("(?:" + expr_to_regex(value) + ")+")
elif kind == "space":
parts.append(SPACE)
elif kind == "stdlib":
Expand Down Expand Up @@ -280,6 +289,8 @@ def expr_to_literal(expr: str) -> str:
parts.append(" ")
elif kind == "opt":
parts.append(expr_to_literal(value))
elif kind == "rep":
parts.append(expr_to_literal(value))
elif kind in ("boundary", "anchor"):
raise ValueError(f"{kind} is not valid in a result expression")
return "".join(parts)
Expand All @@ -301,7 +312,7 @@ def expr_max_length(expr: str) -> int:
total += 1
elif kind == "alt":
total += max(expr_max_length(a) for a in value.split("\x00"))
elif kind == "opt":
elif kind in ("opt", "rep"):
total += expr_max_length(value)
elif kind == "grp":
total += expr_max_length(value)
Expand Down
14 changes: 11 additions & 3 deletions src/interscript/isc.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,11 @@
import re
from pathlib import Path

from .expr import expr_lookbehind, expr_neg_lookbehind, expr_to_regex
from .expr import _BOUNDARY, _WORD, expr_lookbehind, expr_neg_lookbehind, expr_to_regex

# Ruby \w is ASCII-only; the same divergence the expression layer fixes.
_WORD_BOUNDARY = _BOUNDARY
_NON_WORD_BOUNDARY = "(?:(?<=" + _WORD + ")(?=" + _WORD + ")|(?<!" + _WORD + ")(?!" + _WORD + "))"


class IscParseError(ValueError):
Expand Down Expand Up @@ -674,6 +678,8 @@ def _render_item(item: dict, aliases: dict[str, str]) -> str:
return "capture(" + _render_item(item["inner"], aliases) + ")"
if kind == "maybe":
return "maybe(" + _render_item(item["inner"], aliases) + ")"
if kind == "some":
return "some(" + _render_item(item["inner"], aliases) + ")"
if kind == "primitive":
name = item["name"]
if name == "space":
Expand Down Expand Up @@ -738,15 +744,17 @@ def _regex_of(item: dict, aliases: dict[str, str]) -> str:
return "[" + re.escape(item["lo"]) + "-" + re.escape(item["hi"]) + "]"
if kind == "maybe":
return "(?:" + _regex_of(item["inner"], aliases) + ")?"
if kind == "some":
return "(?:" + _regex_of(item["inner"], aliases) + ")+"
if kind == "capture_group":
return "(" + _regex_of(item["inner"], aliases) + ")"
if kind == "capture":
# ref(N) in a from position is a backreference.
return f"\\{item['index']}"
if kind == "primitive":
return {
"boundary": r"\b",
"non_word_boundary": r"\B",
"boundary": _WORD_BOUNDARY,
"non_word_boundary": _NON_WORD_BOUNDARY,
"line_start": "^",
"line_end": "$",
"space": " ",
Expand Down
24 changes: 18 additions & 6 deletions tests/test_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -150,20 +150,23 @@ def test_parallel_selection_matches_ruby_max_length():
assert Engine(tree2).transliterate("abc") == "Y"


def test_boundary_treats_combining_marks_as_word_chars():
"""Ruby's \\b counts combining marks (Arabic diacritics) as word
characters — a word-final rule must not fire when a kasra follows
the hamza carrier (dā'im, not dā'aim)."""
def test_boundary_uses_the_unicode_word_property_like_ruby():
"""Ruby's \\b is Unicode-aware over the Word property, which counts
combining marks as word characters (its \\w is ASCII-only, but \\b
does not follow \\w there). A word-final rule must not fire when a
kasra follows the hamza carrier (dā'im, not dā'aim), and must fire
at a true end of word."""
tree = parse_imp(
'stage {\n parallel {\n'
' sub "ئ" + boundary, "\'a"\n'
' sub "ئ", "\'"\n'
' sub "ِ", "i"\n'
' sub "d", "d"\n }\n}\n'
)
# ئ + kasra: no boundary — the bare-ئ rule fires, not the final one.
# ئ + kasra: the kasra is a Mark, hence a word char — no boundary,
# and the bare-ئ rule fires, not the word-final one.
assert Engine(tree).transliterate("dئِ") == "d'i"
# ئ at a true word end: the boundary rule fires.
# ئ at a true end of word: the boundary rule fires.
assert Engine(tree).transliterate("dئ") == "d'a"


Expand Down Expand Up @@ -291,3 +294,12 @@ def test_primitive_space_result_pads_the_string():
fired: 불국사 -> ᄇulguksa instead of Bulguksa."""
e = _load("moct-kor-Hang-Latn-2000")
assert e.transliterate("불국사") == "Bulguksa"


@pytest.mark.skipif(not MAPS.is_dir(), reason="interscript maps repo not present")
def test_subst_boundary_treats_combining_marks_as_word_chars():
"""un-mar कंगना: the subst-family path used raw \\b, so the क|ं
junction (anusvara is a combining mark) counted as a word boundary
and the schwa-killing rule fired — kaṁganā came out kṁganā."""
e = _load("un-mar-Deva-Latn-2016")
assert e.transliterate("कंगना") == "kaṁganā"
6 changes: 3 additions & 3 deletions tests/test_isc.py
Original file line number Diff line number Diff line change
Expand Up @@ -109,8 +109,8 @@ def test_parse_error_carries_position():


def test_unsupported_construct_raises_by_default():
bad = 'system "x" { stage main { parallel { sub some("a") "b" } } }'
with pytest.raises(UnsupportedConstruct, match="some"):
bad = 'system "x" { stage main { parallel { sub { from "a" to "b" before unresolved_alias } } } }'
with pytest.raises(UnsupportedConstruct, match="unresolved"):
isc_to_tree(bad)


Expand All @@ -127,7 +127,7 @@ def test_skip_mode_records_and_drops():
tree = isc_to_tree(bad, on_unsupported="skip")
assert tree["skipped_unsupported"] == []
tree2 = isc_to_tree(
'system "x" { stage main { sequence { sub some("c") "d" } } }',
'system "x" { stage main { sequence { sub { from "c" to "d" before unresolved_alias } } } }',
on_unsupported="skip",
)
assert tree2["skipped_unsupported"]
Expand Down
Loading