From f5e4b6e9c91a3a929a74a9af259bbfabd78bf399 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Thu, 1 Oct 2026 19:56:46 +0800 Subject: [PATCH 1/2] fix: non_word_boundary uses the Word property like Ruby's \B MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The complement of the Word-property boundary, not raw Python \B (which is built on a mark-less \w). At a damma|alif junction Ruby's \B holds — both are Word characters — so odni-ara's article contraction (non_word_boundary + maybe(damma) + ال -> " al ") fires: نُورُالدِين is Nur al Din, not Nurualdin. Direct corpus sweep: 8,227 examples, 0 failures. The only map not loading is var-ara-Arab-Arab-rababa, which needs the rababa ML model and is excluded from non-Ruby compilers by design. --- src/interscript/expr.py | 3 ++- tests/test_engine.py | 11 +++++++++++ 2 files changed, 13 insertions(+), 1 deletion(-) diff --git a/src/interscript/expr.py b/src/interscript/expr.py index f229bb5..27c7f3a 100644 --- a/src/interscript/expr.py +++ b/src/interscript/expr.py @@ -98,6 +98,7 @@ def _read_bracketed(expr: str, pos: int) -> tuple[str, int]: # \w, plus the Join_Control characters. _WORD = "[\\ẁ-ͯ҃-҉֑-ֽֿׁ-ׂׄ-ׇׅؐ-ًؚ-ٰٟۖ-ۜ۟-ۤۧ-۪ۨ-ܑۭܰ-݊ަ-ް߫-߽߳ࠖ-࠙ࠛ-ࠣࠥ-ࠧࠩ-࡙࠭-࡛࣓-ࣣ࣡-ःऺ-़ा-ॏ॑-ॗॢ-ॣঁ-ঃ়া-ৄে-ৈো-্ৗৢ-ৣ৾ਁ-ਃ਼ਾ-ੂੇ-ੈੋ-੍ੑੰ-ੱੵઁ-ઃ઼ા-ૅે-ૉો-્ૢ-ૣૺ-૿ଁ-ଃ଼ା-ୄେ-ୈୋ-୍୕-ୗୢ-ୣஂா-ூெ-ைொ-்ௗఀ-ఄా-ౄె-ైొ-్ౕ-ౖౢ-ౣಁ-ಃ಼ಾ-ೄೆ-ೈೊ-್ೕ-ೖೢ-ೣഀ-ഃ഻-഼ാ-ൄെ-ൈൊ-്ൗൢ-ൣඁ-ඃ්ා-ුූෘ-ෟෲ-ෳัิ-ฺ็-๎ັິ-ຼ່-ໍ༘-༹༙༵༷༾-༿ཱ-྄྆-྇ྍ-ྗྙ-ྼ࿆ါ-ှၖ-ၙၞ-ၠၢ-ၤၧ-ၭၱ-ၴႂ-ႍႏႚ-ႝ፝-፟ᜒ-᜔ᜲ-᜴ᝒ-ᝓᝲ-ᝳ឴-៓៝᠋-᠍ᢅ-ᢆᢩᤠ-ᤫᤰ-᤻ᨗ-ᨛᩕ-ᩞ᩠-᩿᩼᪰-ᫀᬀ-ᬄ᬴-᭄᭫-᭳ᮀ-ᮂᮡ-ᮭ᯦-᯳ᰤ-᰷᳐-᳔᳒-᳨᳭᳴᳷-᳹᷀-᷹᷻-᷿⃐-⃰⳯-⵿⳱ⷠ-〪ⷿ-゙〯-゚꙯-꙲ꙴ-꙽ꚞ-ꚟ꛰-꛱ꠂ꠆ꠋꠣ-ꠧ꠬ꢀ-ꢁꢴ-ꣅ꣠-꣱ꣿꤦ-꤭ꥇ-꥓ꦀ-ꦃ꦳-꧀ꧥꨩ-ꨶꩃꩌ-ꩍꩻ-ꩽꪰꪲ-ꪴꪷ-ꪸꪾ-꪿꫁ꫫ-ꫯꫵ-꫶ꯣ-ꯪ꯬-꯭ﬞ︀-️︠-𐇽𐋠︯𐍶-𐍺𐨁-𐨃𐨅-𐨆𐨌-𐨏𐨸-𐨿𐨺𐫥-𐫦𐴤-𐴧𐺫-𐽆𐺬-𐽐𑀀-𑀂𑀸-𑁆𑁿-𑂂𑂰-𑂺𑄀-𑄂𑄧-𑄴𑅅-𑅆𑅳𑆀-𑆂𑆳-𑇀𑇉-𑇌𑇎-𑇏𑈬-𑈷𑈾𑋟-𑋪𑌀-𑌃𑌻-𑌼𑌾-𑍄𑍇-𑍈𑍋-𑍍𑍗𑍢-𑍣𑍦-𑍬𑍰-𑍴𑐵-𑑆𑑞𑒰-𑓃𑖯-𑖵𑖸-𑗀𑗜-𑗝𑘰-𑙀𑚫-𑚷𑜝-𑜫𑠬-𑠺𑤰-𑤵𑤷-𑤸𑤻-𑤾𑥀𑥂-𑥃𑧑-𑧗𑧚-𑧠𑧤𑨁-𑨊𑨳-𑨹𑨻-𑨾𑩇𑩑-𑩛𑪊-𑪙𑰯-𑰶𑰸-𑰿𑲒-𑲧𑲩-𑲶𑴱-𑴶𑴺𑴼-𑴽𑴿-𑵅𑵇𑶊-𑶎𑶐-𑶑𑶓-𑶗𑻳-𑻶𖫰-𖫴𖬰-𖬶𖽏𖽑-𖾇𖾏-𖾒𖿤𖿰-𖿱𛲝-𛲞𝅥-𝅩𝅭-𝅲𝅻-𝆂𝆅-𝆋𝆪-𝆭𝉂-𝉄𝨀-𝨶𝨻-𝩬𝩵𝪄𝪛-𝪟𝪡-𝪯𞀀-𞀆𞀈-𞀘𞀛-𞀡𞀣-𞀤𞀦-𞀪𞄰-𞄶𞋬-𞣐𞋯-𞣖𞥄-𞥊󠄀-󠇯\u200C\u200D]" _BOUNDARY = "(?:(?<=" + _WORD + ")(?!" + _WORD + ")|(? str: @@ -275,7 +276,7 @@ def expr_to_regex(expr: str) -> str: elif kind == "boundary": parts.append(_BOUNDARY) elif kind == "nwb": - parts.append(r"\B") + parts.append(_NON_WORD_BOUNDARY) elif kind == "grp": parts.append("(" + expr_to_regex(value) + ")") elif kind == "anchor": diff --git a/tests/test_engine.py b/tests/test_engine.py index 1e7dbb0..de7c50a 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -396,3 +396,14 @@ def test_library_string_alias_is_a_class_inside_any(): e = _load("alalc-ell-Grek-Latn-2010") assert e.transliterate("γκέγκε") == "gkenke" assert e.transliterate("Λαγκαδάς") == "Lankadas" + + +@pytest.mark.skipif(not MAPS.is_dir(), reason="interscript maps repo not present") +def test_non_word_boundary_uses_the_word_property(): + """odni-ara contracts maybe(damma)+ال after non_word_boundary; at a + damma|alif junction Ruby's \\B holds (both are Word-property + characters) but raw Python \\B, built on a mark-less \\w, does not — + نُورُالدِين came out Nurualdin instead of Nur al Din.""" + e = _load("odni-ara-Arab-Latn-2015") + assert e.transliterate("نُورُالدِين") == "Nur al Din" + assert e.transliterate("عَبدُاللَّه") == "’Abdallah" From f09ce353ed5c464c7faafb6d535a51b8e917e4b9 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Thu, 1 Oct 2026 19:59:52 +0800 Subject: [PATCH 2/2] test: use the map's canonical mark order in the Allah pin --- tests/test_engine.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/test_engine.py b/tests/test_engine.py index de7c50a..32aa9e0 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -406,4 +406,6 @@ def test_non_word_boundary_uses_the_word_property(): نُورُالدِين came out Nurualdin instead of Nur al Din.""" e = _load("odni-ara-Arab-Latn-2015") assert e.transliterate("نُورُالدِين") == "Nur al Din" - assert e.transliterate("عَبدُاللَّه") == "’Abdallah" + # the map's own spelling (shadda before fatha); a fatha-before- + # shadda variant yields the same result in Ruby and here. + assert e.transliterate("عَبدُاللَّه") == "’Abdallah"