diff --git a/src/interscript/expr.py b/src/interscript/expr.py index 18cd861..3765f2a 100644 --- a/src/interscript/expr.py +++ b/src/interscript/expr.py @@ -87,14 +87,15 @@ def _read_bracketed(expr: str, pos: int) -> tuple[str, int]: "any_character": ".", } -# Ruby's \b counts combining marks as word characters; Python's \w -# does not (Mn is not alphanumeric). At a hamza-carrier + kasra -# junction Ruby sees no boundary while Python does — word-final rules -# fired wrongly and doubled vowels. Express the boundary as an -# explicit word/non-word transition over a word class that includes -# combining marks (Mnemonic ranges: combining diacritics 0300-036F, -# Arabic diacritics 064B-065F, 0670, and Quranic annotation 06D6-06ED). -_WORD = r"[\w\u0300-\u036F\u064B-\u065F\u0670\u06D6-\u06ED]" +# Ruby's \b is Unicode-aware (the Word property: letters, MARKS, +# digits, connectors) while its \w stays ASCII-only; Python's \b uses +# \w, which excludes combining marks. At a hamza-carrier + kasra or +# क + anusvara junction Ruby therefore sees no boundary where Python +# does, and word-final rules fired wrongly (dā'aim for dā'im, +# kṁganā for kaṁganā). The boundary is expressed explicitly over a +# word class that adds every Mark range (290 ranges, generated) to +# \w, plus the Join_Control characters. +_WORD = "[\\ẁ-ͯ҃-҉֑-ֽֿׁ-ׂׄ-ׇׅؐ-ًؚ-ٰٟۖ-ۜ۟-ۤۧ-۪ۨ-ܑۭܰ-݊ަ-ް߫-߽߳ࠖ-࠙ࠛ-ࠣࠥ-ࠧࠩ-࡙࠭-࡛࣓-ࣣ࣡-ःऺ-़ा-ॏ॑-ॗॢ-ॣঁ-ঃ়া-ৄে-ৈো-্ৗৢ-ৣ৾ਁ-ਃ਼ਾ-ੂੇ-ੈੋ-੍ੑੰ-ੱੵઁ-ઃ઼ા-ૅે-ૉો-્ૢ-ૣૺ-૿ଁ-ଃ଼ା-ୄେ-ୈୋ-୍୕-ୗୢ-ୣஂா-ூெ-ைொ-்ௗఀ-ఄా-ౄె-ైొ-్ౕ-ౖౢ-ౣಁ-ಃ಼ಾ-ೄೆ-ೈೊ-್ೕ-ೖೢ-ೣഀ-ഃ഻-഼ാ-ൄെ-ൈൊ-്ൗൢ-ൣඁ-ඃ්ා-ුූෘ-ෟෲ-ෳัิ-ฺ็-๎ັິ-ຼ່-ໍ༘-༹༙༵༷༾-༿ཱ-྄྆-྇ྍ-ྗྙ-ྼ࿆ါ-ှၖ-ၙၞ-ၠၢ-ၤၧ-ၭၱ-ၴႂ-ႍႏႚ-ႝ፝-፟ᜒ-᜔ᜲ-᜴ᝒ-ᝓᝲ-ᝳ឴-៓៝᠋-᠍ᢅ-ᢆᢩᤠ-ᤫᤰ-᤻ᨗ-ᨛᩕ-ᩞ᩠-᩿᩼᪰-ᫀᬀ-ᬄ᬴-᭄᭫-᭳ᮀ-ᮂᮡ-ᮭ᯦-᯳ᰤ-᰷᳐-᳔᳒-᳨᳭᳴᳷-᳹᷀-᷹᷻-᷿⃐-⃰⳯-⵿⳱ⷠ-〪ⷿ-゙〯-゚꙯-꙲ꙴ-꙽ꚞ-ꚟ꛰-꛱ꠂ꠆ꠋꠣ-ꠧ꠬ꢀ-ꢁꢴ-ꣅ꣠-꣱ꣿꤦ-꤭ꥇ-꥓ꦀ-ꦃ꦳-꧀ꧥꨩ-ꨶꩃꩌ-ꩍꩻ-ꩽꪰꪲ-ꪴꪷ-ꪸꪾ-꪿꫁ꫫ-ꫯꫵ-꫶ꯣ-ꯪ꯬-꯭ﬞ︀-️︠-𐇽𐋠︯𐍶-𐍺𐨁-𐨃𐨅-𐨆𐨌-𐨏𐨸-𐨿𐨺𐫥-𐫦𐴤-𐴧𐺫-𐽆𐺬-𐽐𑀀-𑀂𑀸-𑁆𑁿-𑂂𑂰-𑂺𑄀-𑄂𑄧-𑄴𑅅-𑅆𑅳𑆀-𑆂𑆳-𑇀𑇉-𑇌𑇎-𑇏𑈬-𑈷𑈾𑋟-𑋪𑌀-𑌃𑌻-𑌼𑌾-𑍄𑍇-𑍈𑍋-𑍍𑍗𑍢-𑍣𑍦-𑍬𑍰-𑍴𑐵-𑑆𑑞𑒰-𑓃𑖯-𑖵𑖸-𑗀𑗜-𑗝𑘰-𑙀𑚫-𑚷𑜝-𑜫𑠬-𑠺𑤰-𑤵𑤷-𑤸𑤻-𑤾𑥀𑥂-𑥃𑧑-𑧗𑧚-𑧠𑧤𑨁-𑨊𑨳-𑨹𑨻-𑨾𑩇𑩑-𑩛𑪊-𑪙𑰯-𑰶𑰸-𑰿𑲒-𑲧𑲩-𑲶𑴱-𑴶𑴺𑴼-𑴽𑴿-𑵅𑵇𑶊-𑶎𑶐-𑶑𑶓-𑶗𑻳-𑻶𖫰-𖫴𖬰-𖬶𖽏𖽑-𖾇𖾏-𖾒𖿤𖿰-𖿱𛲝-𛲞𝅥-𝅩𝅭-𝅲𝅻-𝆂𝆅-𝆋𝆪-𝆭𝉂-𝉄𝨀-𝨶𝨻-𝩬𝩵𝪄𝪛-𝪟𝪡-𝪯𞀀-𞀆𞀈-𞀘𞀛-𞀡𞀣-𞀤𞀦-𞀪𞄰-𞄶𞋬-𞣐𞋯-𞣖𞥄-𞥊󠄀-󠇯\u200C\u200D]" _BOUNDARY = "(?:(?<=" + _WORD + ")(?!" + _WORD + ")|(? str: _ANY_LIST = re.compile(r"any\(\s*\[") _MAYBE = re.compile(r"maybe\(\s*") +_SOME = re.compile(r"some\(\s*") def _read_parenthesized(expr: str, pos: int) -> tuple[str, int]: @@ -139,6 +141,11 @@ def _scan(expr: str, want: str): if expr[pos].isspace(): pos += 1 continue + if _SOME.match(expr, pos): + paren = expr.find("(", pos) + inner, pos = _read_parenthesized(expr, paren) + out.append(("rep", inner)) + continue if _MAYBE.match(expr, pos): paren = expr.find("(", pos) inner, pos = _read_parenthesized(expr, paren) @@ -252,6 +259,8 @@ def expr_to_regex(expr: str) -> str: parts.append("(?:" + "|".join(expr_to_regex(a) for a in alts) + ")") elif kind == "opt": parts.append("(?:" + expr_to_regex(value) + ")?") + elif kind == "rep": + parts.append("(?:" + expr_to_regex(value) + ")+") elif kind == "space": parts.append(SPACE) elif kind == "stdlib": @@ -280,6 +289,8 @@ def expr_to_literal(expr: str) -> str: parts.append(" ") elif kind == "opt": parts.append(expr_to_literal(value)) + elif kind == "rep": + parts.append(expr_to_literal(value)) elif kind in ("boundary", "anchor"): raise ValueError(f"{kind} is not valid in a result expression") return "".join(parts) @@ -301,7 +312,7 @@ def expr_max_length(expr: str) -> int: total += 1 elif kind == "alt": total += max(expr_max_length(a) for a in value.split("\x00")) - elif kind == "opt": + elif kind in ("opt", "rep"): total += expr_max_length(value) elif kind == "grp": total += expr_max_length(value) diff --git a/src/interscript/isc.py b/src/interscript/isc.py index 21e67bf..6d3535b 100644 --- a/src/interscript/isc.py +++ b/src/interscript/isc.py @@ -22,7 +22,11 @@ import re from pathlib import Path -from .expr import expr_lookbehind, expr_neg_lookbehind, expr_to_regex +from .expr import _BOUNDARY, _WORD, expr_lookbehind, expr_neg_lookbehind, expr_to_regex + +# Ruby \w is ASCII-only; the same divergence the expression layer fixes. +_WORD_BOUNDARY = _BOUNDARY +_NON_WORD_BOUNDARY = "(?:(?<=" + _WORD + ")(?=" + _WORD + ")|(? str: return "capture(" + _render_item(item["inner"], aliases) + ")" if kind == "maybe": return "maybe(" + _render_item(item["inner"], aliases) + ")" + if kind == "some": + return "some(" + _render_item(item["inner"], aliases) + ")" if kind == "primitive": name = item["name"] if name == "space": @@ -738,6 +744,8 @@ def _regex_of(item: dict, aliases: dict[str, str]) -> str: return "[" + re.escape(item["lo"]) + "-" + re.escape(item["hi"]) + "]" if kind == "maybe": return "(?:" + _regex_of(item["inner"], aliases) + ")?" + if kind == "some": + return "(?:" + _regex_of(item["inner"], aliases) + ")+" if kind == "capture_group": return "(" + _regex_of(item["inner"], aliases) + ")" if kind == "capture": @@ -745,8 +753,8 @@ def _regex_of(item: dict, aliases: dict[str, str]) -> str: return f"\\{item['index']}" if kind == "primitive": return { - "boundary": r"\b", - "non_word_boundary": r"\B", + "boundary": _WORD_BOUNDARY, + "non_word_boundary": _NON_WORD_BOUNDARY, "line_start": "^", "line_end": "$", "space": " ", diff --git a/tests/test_engine.py b/tests/test_engine.py index 9ce9f01..7624285 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -150,10 +150,12 @@ def test_parallel_selection_matches_ruby_max_length(): assert Engine(tree2).transliterate("abc") == "Y" -def test_boundary_treats_combining_marks_as_word_chars(): - """Ruby's \\b counts combining marks (Arabic diacritics) as word - characters — a word-final rule must not fire when a kasra follows - the hamza carrier (dā'im, not dā'aim).""" +def test_boundary_uses_the_unicode_word_property_like_ruby(): + """Ruby's \\b is Unicode-aware over the Word property, which counts + combining marks as word characters (its \\w is ASCII-only, but \\b + does not follow \\w there). A word-final rule must not fire when a + kasra follows the hamza carrier (dā'im, not dā'aim), and must fire + at a true end of word.""" tree = parse_imp( 'stage {\n parallel {\n' ' sub "ئ" + boundary, "\'a"\n' @@ -161,9 +163,10 @@ def test_boundary_treats_combining_marks_as_word_chars(): ' sub "ِ", "i"\n' ' sub "d", "d"\n }\n}\n' ) - # ئ + kasra: no boundary — the bare-ئ rule fires, not the final one. + # ئ + kasra: the kasra is a Mark, hence a word char — no boundary, + # and the bare-ئ rule fires, not the word-final one. assert Engine(tree).transliterate("dئِ") == "d'i" - # ئ at a true word end: the boundary rule fires. + # ئ at a true end of word: the boundary rule fires. assert Engine(tree).transliterate("dئ") == "d'a" @@ -291,3 +294,12 @@ def test_primitive_space_result_pads_the_string(): fired: 불국사 -> ᄇulguksa instead of Bulguksa.""" e = _load("moct-kor-Hang-Latn-2000") assert e.transliterate("불국사") == "Bulguksa" + + +@pytest.mark.skipif(not MAPS.is_dir(), reason="interscript maps repo not present") +def test_subst_boundary_treats_combining_marks_as_word_chars(): + """un-mar कंगना: the subst-family path used raw \\b, so the क|ं + junction (anusvara is a combining mark) counted as a word boundary + and the schwa-killing rule fired — kaṁganā came out kṁganā.""" + e = _load("un-mar-Deva-Latn-2016") + assert e.transliterate("कंगना") == "kaṁganā" diff --git a/tests/test_isc.py b/tests/test_isc.py index 74a416c..0bb8b0d 100644 --- a/tests/test_isc.py +++ b/tests/test_isc.py @@ -109,8 +109,8 @@ def test_parse_error_carries_position(): def test_unsupported_construct_raises_by_default(): - bad = 'system "x" { stage main { parallel { sub some("a") "b" } } }' - with pytest.raises(UnsupportedConstruct, match="some"): + bad = 'system "x" { stage main { parallel { sub { from "a" to "b" before unresolved_alias } } } }' + with pytest.raises(UnsupportedConstruct, match="unresolved"): isc_to_tree(bad) @@ -127,7 +127,7 @@ def test_skip_mode_records_and_drops(): tree = isc_to_tree(bad, on_unsupported="skip") assert tree["skipped_unsupported"] == [] tree2 = isc_to_tree( - 'system "x" { stage main { sequence { sub some("c") "d" } } }', + 'system "x" { stage main { sequence { sub { from "c" to "d" before unresolved_alias } } } }', on_unsupported="skip", ) assert tree2["skipped_unsupported"]