From cc0c7b746c0cddf9450edb5375d12b9c1462476c Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Wed, 30 Sep 2026 09:15:22 +0800 Subject: [PATCH 01/10] =?UTF-8?q?fix:=20any-list=20entries=20stay=20Items?= =?UTF-8?q?=20=E2=80=94=20kills=20the=20char-set=20inspect=20leak?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit rule(list:) built Items::Set.from_strings(arr.map(&:to_s)); stringifying an any([...]) list baked object inspects into the char set, so maps whose sets contained primitives, alias refs or multi-char entries compiled a garbage class (iso-mal's "പ്പം ഹ" cases). Keep the Items as they are — convert_set already normalizes each entry kind. With this, the env-gated ISC corpus sweep runs end-to-end on the 289-map corpus. It is NOT green yet: the sweep now measures a real conformance gap (case/capitalization post-rules, escape decoding, cluster ordering) that the following commits close. --- .github/workflows/rake.yml | 4 ++++ lib/interscript/isc/node_adapter.rb | 2 ++ lib/interscript/isc/transform.rb | 4 +++- spec/interscript/isc/node_adapter_spec.rb | 17 +++++++++++++++++ spec/interscript_spec.rb | 10 +++++++++- spec/map_name_and_metadata_spec.rb | 12 +++++++++++- 6 files changed, 46 insertions(+), 3 deletions(-) diff --git a/.github/workflows/rake.yml b/.github/workflows/rake.yml index c15a542d..d70d609e 100644 --- a/.github/workflows/rake.yml +++ b/.github/workflows/rake.yml @@ -18,6 +18,10 @@ jobs: env: BUNDLE_WITHOUT: "secryst" SKIP_JS: "1" + # The ISC corpus sweep: every map's own tests through every + # compiler. The ../maps sibling is on the load path via the + # Gemfile path dependency. + ISC_SWEEP: "1" steps: - name: Checkout monorepo diff --git a/lib/interscript/isc/node_adapter.rb b/lib/interscript/isc/node_adapter.rb index c2cca5eb..4f0baefb 100644 --- a/lib/interscript/isc/node_adapter.rb +++ b/lib/interscript/isc/node_adapter.rb @@ -209,6 +209,8 @@ def convert_set(set) case c when ::String Interscript::Node::Item::String.new(c) + when Items::StringValue + Interscript::Node::Item::String.new(c.value) when Items::Primitive convert_primitive(c) when Items::AliasRef diff --git a/lib/interscript/isc/transform.rb b/lib/interscript/isc/transform.rb index fcf15dad..601b9a31 100644 --- a/lib/interscript/isc/transform.rb +++ b/lib/interscript/isc/transform.rb @@ -75,8 +75,10 @@ class Transform < Parslet::Transform Items::Range.new(lo.to_s, hi.to_s) end rule(single: simple(:s)) { Items::Set.from_string(s.to_s) } + # List entries are already Items (strings, primitives, alias refs). + # Stringifying here baked an object inspect into the char set. rule(list: sequence(:arr)) do - Items::Set.from_strings(arr.map(&:to_s)) + Items::Set.new(arr) end rule(any: subtree(:h)) { h } diff --git a/spec/interscript/isc/node_adapter_spec.rb b/spec/interscript/isc/node_adapter_spec.rb index 301e8a83..38a09601 100644 --- a/spec/interscript/isc/node_adapter_spec.rb +++ b/spec/interscript/isc/node_adapter_spec.rb @@ -263,3 +263,20 @@ def parse_and_adapt(src) expect(node.name).to eq("X:a-b:C-D:1") end end + +RSpec.describe "NodeAdapter any-list constraints" do + it "keeps primitives as aliases, never their inspect" do + tree = Interscript::Isc::Parser.parse( + 'system "x" { stage main { sub { from "ം" to "m" after any([boundary, "‌", "‍"]) } } }' + ) + doc = Interscript::Isc::DocumentBuilder.build(tree) + node = Interscript::Isc::NodeAdapter.to_interscript_node(doc) + constraint = node.stages[:main].children.first.after + stage = Interscript::Interpreter::Stage.new(node, "") + re = stage.send(:build_regexp, node.stages[:main].children.first) + # The constraint must match boundary positions — an inspect leak + # turns the lookahead into a garbage char class. + expect(re).not_to include("Primitive") + expect("പ്പം ഹ").to match(Regexp.new(re)) + end +end diff --git a/spec/interscript_spec.rb b/spec/interscript_spec.rb index 1f022067..09fb7b97 100644 --- a/spec/interscript_spec.rb +++ b/spec/interscript_spec.rb @@ -7,8 +7,16 @@ # per-map failures are expected until the ISC-era conformance work # completes — the sweep exists to MEASURE that gap, gated so CI stays # green while it closes. +# The sweep honors INTERSCRIPT_MAPS_PATH (like the corpus specs) and +# enumerates through the load path — matching what locate() resolves. +# Bare maps() would enumerate the installed gem's frozen 2.4.x .imp +# corpus wherever the maps checkout isn't the active gem. maps = if ENV["ISC_SWEEP"] - Interscript.maps(basename: false, select: mask) + sweep_root = ENV.fetch("INTERSCRIPT_MAPS_PATH", "../maps/maps") + unless Interscript.load_path.first == File.expand_path(sweep_root) + Interscript.load_path.unshift(File.expand_path(sweep_root)) + end + Interscript.maps(basename: false, select: mask, load_path: true) else legacy_maps(select: mask) end diff --git a/spec/map_name_and_metadata_spec.rb b/spec/map_name_and_metadata_spec.rb index e11dece2..6510e69e 100644 --- a/spec/map_name_and_metadata_spec.rb +++ b/spec/map_name_and_metadata_spec.rb @@ -5,7 +5,17 @@ RSpec.describe "map names and metadata" do valid_authcodes = YAML.load_file(__dir__ + "/authority_codes.yaml").keys - legacy_maps.each do |n| + maps = + if ENV["ISC_SWEEP"] + sweep_root = ENV.fetch("INTERSCRIPT_MAPS_PATH", "../maps/maps") + unless Interscript.load_path.first == File.expand_path(sweep_root) + Interscript.load_path.unshift(File.expand_path(sweep_root)) + end + Interscript.maps(load_path: true, libraries: false) + else + legacy_maps + end + maps.each do |n| context n do parts = n.split("-", 5) authcode, lang, source_script, target_script, id = parts From 3988a0b8d89164932e3cb137e6997afe164c465e Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Wed, 30 Sep 2026 10:17:35 +0800 Subject: [PATCH 02/10] fix: cache parsed ISC documents in DSL.parse MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The .isc branch bypassed @cache, so every transliterate() call re-ran the full Parslet parse — a 700 KB map costs ~30 s per call and blew the 5 s per-example test timeout. The .imp path already caches; parity. --- lib/interscript/dsl.rb | 4 +++- spec/maps_listing_spec.rb | 17 +++++++++++++++++ 2 files changed, 20 insertions(+), 1 deletion(-) diff --git a/lib/interscript/dsl.rb b/lib/interscript/dsl.rb index 07a89a3c..b085b021 100644 --- a/lib/interscript/dsl.rb +++ b/lib/interscript/dsl.rb @@ -40,7 +40,9 @@ def self.parse(map_name, reverse: true) end end # ISC documents route through the ISC parser, not the .imp DSL. - return Interscript::Compiler.parse_isc(path) if path.end_with?(".isc") + # The parse is expensive (Parslet over the whole map) — cache it like + # the .imp path does, or every transliterate call re-parses the map. + return @cache[map_name] = Interscript::Compiler.parse_isc(path) if path.end_with?(".isc") library = path.end_with?(".iml") diff --git a/spec/maps_listing_spec.rb b/spec/maps_listing_spec.rb index 99a59694..8c2f7b44 100644 --- a/spec/maps_listing_spec.rb +++ b/spec/maps_listing_spec.rb @@ -53,3 +53,20 @@ def with_load_path(dir) end end end + +RSpec.describe "Interscript.parse caching for ISC documents" do + it "returns the cached document — repeated parses must not re-run the ISC parser" do + maps = File.expand_path(MAPS) + skip "maps checkout not present" unless File.file?(File.expand_path("alalc-aze-Arab-Latn-1997.isc", maps)) + + added = Interscript.load_path.first != maps + Interscript.load_path.unshift(maps) if added + begin + first = Interscript.parse("alalc-aze-Arab-Latn-1997") + second = Interscript.parse("alalc-aze-Arab-Latn-1997") + expect(first.equal?(second)).to be(true) + ensure + Interscript.load_path.delete_at(0) if added + end + end +end From 472b658b40f3a522efea120004bfee40791f8329 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Wed, 30 Sep 2026 10:17:35 +0800 Subject: [PATCH 03/10] fix: keep ISC ranges as native Any ranges MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Expanding Items::Range into an Array of single-char Strings iterated Ruby String#succ (a, b, … z, aa, ab …), which never reaches non-ASCII codepoints — ranges like any("a".."\uFFFF") matched no non-ASCII letters, silently dropping the corpus's word-capitalization post-rules (73% of the sweep failures were case-only). Both runtimes compile Any(Range) to a codepoint character class. --- lib/interscript/isc/node_adapter.rb | 10 ++++++--- spec/interscript/isc/node_adapter_spec.rb | 26 +++++++++++++++++++++++ 2 files changed, 33 insertions(+), 3 deletions(-) diff --git a/lib/interscript/isc/node_adapter.rb b/lib/interscript/isc/node_adapter.rb index 4f0baefb..378c9f7c 100644 --- a/lib/interscript/isc/node_adapter.rb +++ b/lib/interscript/isc/node_adapter.rb @@ -174,9 +174,13 @@ def convert_item(item) when Items::Some Interscript::Node::Item::Some.new(convert_item(item.inner)) when Items::Range - Interscript::Node::Item::Any.new( - (item.lo..item.hi).map { |c| Interscript::Node::Item::String.new(c) } - ) + # Keep the range native: both runtimes compile Any(Range) to a + # codepoint character class [lo-hi]. Expanding it to an Array + # iterates Ruby String#succ (a, b, … z, aa, ab …), which never + # reaches non-ASCII codepoints and blows up the node size. + lo = item.lo.is_a?(Items::StringValue) ? item.lo.value : item.lo + hi = item.hi.is_a?(Items::StringValue) ? item.hi.value : item.hi + Interscript::Node::Item::Any.new(lo..hi) when Items::Set convert_set(item) else diff --git a/spec/interscript/isc/node_adapter_spec.rb b/spec/interscript/isc/node_adapter_spec.rb index 38a09601..2b784f74 100644 --- a/spec/interscript/isc/node_adapter_spec.rb +++ b/spec/interscript/isc/node_adapter_spec.rb @@ -264,6 +264,32 @@ def parse_and_adapt(src) end end +RSpec.describe "NodeAdapter range handling" do + it "keeps ISC ranges as native Any ranges — codepoint semantics, not string-succ expansion" do + src = <<~'ISC' + system "TEST:aze-Arab:Latn:2026" { + metadata { name "T" } + stage main { + parallel { sub "q" "k" } + sub { from any("a".."￿") to upcase before boundary } + } + } + ISC + node = Interscript::Isc::NodeAdapter.to_interscript_node( + Interscript::Isc::DocumentBuilder.build(Interscript::Isc::Parser.parse(src)) + ) + upcase_rule = node.stages[:main].children + .select { |c| c.is_a?(Interscript::Node::Rule::Sub) } + .find { |r| r.to == :upcase } + # A Ruby String range expands via String#succ (a, b, ..., z, aa, ab …) + # which never reaches non-ASCII codepoints — the range must stay a range. + expect(upcase_rule.from.value).to be_a(Range) + interp = Interscript::Interpreter.new + interp.compile(node) + expect(interp.call("īş")).to eq("Īş") + end +end + RSpec.describe "NodeAdapter any-list constraints" do it "keeps primitives as aliases, never their inspect" do tree = Interscript::Isc::Parser.parse( From 3ee114616bf29fff6088b514b8b5fd045a7336d7 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Wed, 30 Sep 2026 10:34:59 +0800 Subject: [PATCH 04/10] fix: decode escape sequences when normalizing ISC strings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Normalizer folded :string captures itself and only read the :char key — escape fragments ({dquote:}, {backslash:}, {newline:}, {unicode:}) have no :char key and every escaped character silently vanished (bas-rus "pod\"ezd" tests, gost's hard-sign-to-quote rules, \uXXXX decoding). The Transform already had the correct fold; share it via Transform.decode_string_parts. --- lib/interscript/isc/normalizer.rb | 10 ++-- lib/interscript/isc/transform.rb | 58 ++++++++++++----------- spec/interscript/isc/node_adapter_spec.rb | 21 ++++++++ 3 files changed, 58 insertions(+), 31 deletions(-) diff --git a/lib/interscript/isc/normalizer.rb b/lib/interscript/isc/normalizer.rb index b5e44fd8..8fe2dc71 100644 --- a/lib/interscript/isc/normalizer.rb +++ b/lib/interscript/isc/normalizer.rb @@ -36,10 +36,12 @@ def normalize def scalar(node) return node if node.is_a?(::String) || node.nil? if node.key?(:string) - # Parts are {char:} hashes in the general grammar; some - # constructions (e.g. the rababa directive) capture bare - # slices instead. - Array(node[:string]).map { |part| part.is_a?(Hash) ? part[:char].to_s : part.to_s }.join + # Parts are {char:} hashes and escape fragments ({dquote:}, + # {unicode:}, …) in the general grammar; some constructions + # (e.g. the rababa directive) capture bare slices instead. + # Escapes decode via the shared fold — folding only :char here + # silently dropped every escaped character. + Interscript::Isc::Transform.decode_string_parts(node[:string]) elsif node.key?(:identifier) node[:identifier].to_s elsif node.key?(:raw) diff --git a/lib/interscript/isc/transform.rb b/lib/interscript/isc/transform.rb index 601b9a31..ea1b36ca 100644 --- a/lib/interscript/isc/transform.rb +++ b/lib/interscript/isc/transform.rb @@ -16,33 +16,7 @@ class Transform < Parslet::Transform # StringValue. rule(string: simple(:s)) { Items::StringValue.new(s.to_s) } rule(string: sequence(:parts)) do - combined = parts.map do |p| - case p - when Hash - # Escape sequence fragment: e.g. {newline: "n"}, {unicode: "1234"}, - # {dquote: '"'}, {char: "a"} - if p.key?(:char) - p[:char].to_s - elsif p.key?(:newline) - "\n" - elsif p.key?(:carriage_return) - "\r" - elsif p.key?(:tab) - "\t" - elsif p.key?(:dquote) - '"' - elsif p.key?(:backslash) - "\\" - elsif p.key?(:unicode) - [p[:unicode].to_s.to_i(16)].pack("U") - else - p.to_s - end - else - p.to_s - end - end.join - Items::StringValue.new(combined) + Items::StringValue.new(Transform.decode_string_parts(parts)) end rule(char: simple(:c)) { c.to_s } @@ -71,6 +45,36 @@ class Transform < Parslet::Transform hex.to_s end + # Shared decoder for the pieces of a :string capture — used by the + # transform rules above and by the Normalizer's scalar folding. A + # fragment without an escape key is a raw character slice. + def self.decode_string_parts(parts) + Array(parts).map do |p| + case p + when Hash + if p.key?(:char) + p[:char].to_s + elsif p.key?(:newline) + "\n" + elsif p.key?(:carriage_return) + "\r" + elsif p.key?(:tab) + "\t" + elsif p.key?(:dquote) + '"' + elsif p.key?(:backslash) + "\\" + elsif p.key?(:unicode) + [p[:unicode].to_s.to_i(16)].pack("U") + else + p.to_s + end + else + p.to_s + end + end.join + end + rule(lo: simple(:lo), hi: simple(:hi)) do Items::Range.new(lo.to_s, hi.to_s) end diff --git a/spec/interscript/isc/node_adapter_spec.rb b/spec/interscript/isc/node_adapter_spec.rb index 2b784f74..8e99d767 100644 --- a/spec/interscript/isc/node_adapter_spec.rb +++ b/spec/interscript/isc/node_adapter_spec.rb @@ -306,3 +306,24 @@ def parse_and_adapt(src) expect("പ്പം ഹ").to match(Regexp.new(re)) end end + +RSpec.describe "NodeAdapter string escapes" do + it "decodes escape sequences in test strings" do + src = <<~'ISC' + system "T:a-b:C-D:1" { + metadata { name "T" } + tests { + "pod\"ezd" -> "p\"ezd" + "a\tb\nc" -> "déjà" + } + stage main { sub "a" "b" } + } + ISC + node = Interscript::Isc::NodeAdapter.to_interscript_node( + Interscript::Isc::DocumentBuilder.build(Interscript::Isc::Parser.parse(src)) + ) + tests = node.tests.data + expect(tests[0]).to eq(['pod"ezd', 'p"ezd']) + expect(tests[1]).to eq(["a\tb\nc", "déjà"]) + end +end From fb12aa27f66e1b0bc09e77ab563bf66ed0a8aaf8 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Wed, 30 Sep 2026 10:37:31 +0800 Subject: [PATCH 05/10] ci: keep the ISC conformance sweep opt-in MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The sweep is now runnable end-to-end (parse cache, native ranges, escape decoding) but still measures a real conformance gap across the corpus — enabling it in CI would go red. It stays env-gated for local tracking; flip it on once it reaches zero. --- .github/workflows/rake.yml | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/.github/workflows/rake.yml b/.github/workflows/rake.yml index d70d609e..8b9e1329 100644 --- a/.github/workflows/rake.yml +++ b/.github/workflows/rake.yml @@ -18,10 +18,12 @@ jobs: env: BUNDLE_WITHOUT: "secryst" SKIP_JS: "1" - # The ISC corpus sweep: every map's own tests through every - # compiler. The ../maps sibling is on the load path via the - # Gemfile path dependency. - ISC_SWEEP: "1" + # The ISC corpus sweep (ISC_SWEEP=1) stays opt-in until the + # conformance gap it measures is closed: it currently exposes real + # per-map mismatches across the corpus. Run it locally with + # ISC_SWEEP=1 INTERSCRIPT_MAPS_PATH=../maps/maps \ + # rspec spec/interscript_spec.rb + # to track that gap; flip it on here once it goes green. steps: - name: Checkout monorepo From e461dc278be0435e116398168083882b605e20f5 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Wed, 30 Sep 2026 10:41:04 +0800 Subject: [PATCH 06/10] style: drop unused assignment flagged by standardrb --- spec/interscript/isc/node_adapter_spec.rb | 1 - 1 file changed, 1 deletion(-) diff --git a/spec/interscript/isc/node_adapter_spec.rb b/spec/interscript/isc/node_adapter_spec.rb index 8e99d767..0b13859c 100644 --- a/spec/interscript/isc/node_adapter_spec.rb +++ b/spec/interscript/isc/node_adapter_spec.rb @@ -297,7 +297,6 @@ def parse_and_adapt(src) ) doc = Interscript::Isc::DocumentBuilder.build(tree) node = Interscript::Isc::NodeAdapter.to_interscript_node(doc) - constraint = node.stages[:main].children.first.after stage = Interscript::Interpreter::Stage.new(node, "") re = stage.send(:build_regexp, node.stages[:main].children.first) # The constraint must match boundary positions — an inspect leak From cf38afa8ad1e92ceb254986d152236fb653996e5 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Wed, 30 Sep 2026 10:41:46 +0800 Subject: [PATCH 07/10] style: unquote heredoc without escapes --- spec/interscript/isc/node_adapter_spec.rb | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/spec/interscript/isc/node_adapter_spec.rb b/spec/interscript/isc/node_adapter_spec.rb index 0b13859c..b73c47e7 100644 --- a/spec/interscript/isc/node_adapter_spec.rb +++ b/spec/interscript/isc/node_adapter_spec.rb @@ -266,7 +266,7 @@ def parse_and_adapt(src) RSpec.describe "NodeAdapter range handling" do it "keeps ISC ranges as native Any ranges — codepoint semantics, not string-succ expansion" do - src = <<~'ISC' + src = <<~ISC system "TEST:aze-Arab:Latn:2026" { metadata { name "T" } stage main { From d2ef5a9fa667e3a70a83f203c8b7659fdcdd681e Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Wed, 30 Sep 2026 10:55:13 +0800 Subject: [PATCH 08/10] fix: route Compiler.call through DSL.parse for ISC maps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Compiler.call called parse_isc directly for .isc paths, re-running a ~30 s Parslet parse on every compiler instantiation — the spec's shared compile cache never got a hit on the first call per compiler, so the first example of every large map timed out. DSL.parse locates, dispatches and caches; one entry point, one cache. --- lib/interscript/compiler.rb | 16 ++++------------ spec/maps_listing_spec.rb | 17 +++++++++++++++++ 2 files changed, 21 insertions(+), 12 deletions(-) diff --git a/lib/interscript/compiler.rb b/lib/interscript/compiler.rb index fe0753ab..6b1a799a 100644 --- a/lib/interscript/compiler.rb +++ b/lib/interscript/compiler.rb @@ -9,18 +9,10 @@ class Interscript::Compiler attr_accessor :code def self.call(map, **kwargs) - if String === map - path = begin - Interscript.locate(map) - rescue - nil - end - map = if path&.end_with?(".isc") - parse_isc(path) - else - Interscript::DSL.parse(map) - end - end + # DSL.parse locates, dispatches .isc and caches the parsed document — + # calling parse_isc here re-ran a ~30 s Parslet parse on every + # compiler instantiation. + map = Interscript::DSL.parse(map) if String === map compiler = new compiler.compile(map, **kwargs) compiler diff --git a/spec/maps_listing_spec.rb b/spec/maps_listing_spec.rb index 8c2f7b44..c0d91119 100644 --- a/spec/maps_listing_spec.rb +++ b/spec/maps_listing_spec.rb @@ -70,3 +70,20 @@ def with_load_path(dir) end end end + +RSpec.describe "Compiler.call on ISC maps" do + it "reuses the cached document — repeated calls must not re-run the ISC parser" do + maps = File.expand_path(MAPS) + skip "maps checkout not present" unless File.file?(File.expand_path("alalc-aze-Arab-Latn-1997.isc", maps)) + + added = Interscript.load_path.first != maps + Interscript.load_path.unshift(maps) if added + begin + a = Interscript::Interpreter.call("alalc-aze-Arab-Latn-1997") + b = Interscript::Interpreter.call("alalc-aze-Arab-Latn-1997") + expect(a.map.equal?(b.map)).to be(true) + ensure + Interscript.load_path.delete_at(0) if added + end + end +end From 72446e892b77bfcc1f798ffe30ead2782154e4af Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Wed, 30 Sep 2026 14:27:57 +0800 Subject: [PATCH 09/10] fix: alias refs in any() keep set semantics like the legacy DSL MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bare identifiers inside any(...) parsed as :alias_ref, which no transform rule matched — the hash degraded downstream into the debug string "[:alias_ref, \"greek\"]" baked into constraint regexps, so every context-gated rule silently never fired (Greek velar assimilation γ→n, ta-marbuta contexts). Two parts: - transform rules for :alias_ref (list entries and single any() args) - single-alias any() args stay a Set; imported aliases compile to the vacuous fragment exactly as the legacy runtime's Any(nil) does — resolving them to the charset string would bake a 1000-char literal into a lookbehind that never matches --- lib/interscript/isc/node_adapter.rb | 9 ++++++- lib/interscript/isc/transform.rb | 13 ++++++++- spec/interscript/isc/node_adapter_spec.rb | 32 +++++++++++++++++++++++ 3 files changed, 52 insertions(+), 2 deletions(-) diff --git a/lib/interscript/isc/node_adapter.rb b/lib/interscript/isc/node_adapter.rb index 378c9f7c..2110c940 100644 --- a/lib/interscript/isc/node_adapter.rb +++ b/lib/interscript/isc/node_adapter.rb @@ -218,7 +218,14 @@ def convert_set(set) when Items::Primitive convert_primitive(c) when Items::AliasRef - convert_alias_ref(c) + # Legacy parity: a bare alias inside any(...) compiles to the + # Stdlib value (Any(nil) there for imported aliases). Keeping + # the Alias item resolves to the imported charset *string* at + # build time, baking a 1000-char literal into the constraint + # regexp — a lookbehind that never matches. + name = convert_alias_ref(c).name + val = Interscript::Stdlib::ALIASES[name] + val ? Interscript::Node::Item::Any.new(val) : Interscript::Node::Item::String.new("") else convert_item(c) end diff --git a/lib/interscript/isc/transform.rb b/lib/interscript/isc/transform.rb index ea1b36ca..c988dafa 100644 --- a/lib/interscript/isc/transform.rb +++ b/lib/interscript/isc/transform.rb @@ -29,6 +29,11 @@ class Transform < Parslet::Transform Items::AliasRef.new(q.to_s, map: n.to_s) } rule(alias: simple(:n)) { Items::AliasRef.new(n.to_s) } + # Bare identifiers inside any(...) lists parse as :alias_ref (the + # :alias key only covers direct item position). Unmatched, the hash + # degraded downstream into the debug string "[:alias_ref, \"name\"]". + rule(alias_ref: {identifier: simple(:n)}) { Items::AliasRef.new(n.to_s) } + rule(alias_ref: simple(:n)) { Items::AliasRef.new(n.to_s) } rule(ref: subtree(:h)) { Items::Capture.new(h[:digit].to_s.to_i) } rule(capture_inner: subtree(:inner)) { Items::CaptureGroup.new(Interscript::Isc::Transform.materialize_item(inner)) } rule(maybe_inner: subtree(:inner)) { Items::Maybe.new(Interscript::Isc::Transform.materialize_item(inner)) } @@ -84,7 +89,13 @@ def self.decode_string_parts(parts) rule(list: sequence(:arr)) do Items::Set.new(arr) end - rule(any: subtree(:h)) { h } + rule(any: subtree(:h)) do + # A single alias argument keeps set semantics: the legacy runtime + # builds Any(Alias) which resolves through Stdlib (Any(nil) for + # imported aliases). Unwrapped, the bare Alias resolves to the + # imported charset string and compiles as a literal. + h.is_a?(Items::AliasRef) ? Items::Set.new([h]) : h + end rule(concatenation: subtree(:parts)) do Items::Concat.from_parts(Array(parts)) diff --git a/spec/interscript/isc/node_adapter_spec.rb b/spec/interscript/isc/node_adapter_spec.rb index b73c47e7..5941b2d1 100644 --- a/spec/interscript/isc/node_adapter_spec.rb +++ b/spec/interscript/isc/node_adapter_spec.rb @@ -326,3 +326,35 @@ def parse_and_adapt(src) expect(tests[1]).to eq(["a\tb\nc", "déjà"]) end end + +RSpec.describe "NodeAdapter alias refs in constraints" do + it "compiles bare alias refs in sets like the legacy DSL — no debug dump, no literal-charset lookbehind" do + src = <<~ISC + system "t:g-c:C-D:1" { + metadata { name "T" } + stage main { + sub { + from "γ" + to "n" + before any(greek) + after any("κΚ") + any(greek) + not_before boundary + } + } + } + ISC + node = Interscript::Isc::NodeAdapter.to_interscript_node( + Interscript::Isc::DocumentBuilder.build(Interscript::Isc::Parser.parse(src)) + ) + rule = node.stages[:main].children.first + expect(rule.before).to be_a(Interscript::Node::Item::Any) + expect(rule.before.inspect).not_to include("alias_ref") + expect(rule.after.inspect).not_to include("alias_ref") + # The legacy runtime compiles an imported alias in a set to a vacuous + # fragment (Any(nil)); resolving it to the charset *string* would bake + # a 1000-char literal into the lookbehind and never match. + interp = Interscript::Interpreter.new + interp.compile(node) + expect(interp.call("αγκα")).to eq("αnκα") + end +end From 96cd6c7e14fc2eba9d3baba5f1dbff611ea95605 Mon Sep 17 00:00:00 2001 From: Ronald Tse Date: Wed, 30 Sep 2026 14:50:39 +0800 Subject: [PATCH 10/10] fix: make ISC documents addressable like .imp maps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two parity gaps with the legacy DSL, both fatal to Compiler::Ruby: - Document#name now derives from the file name (locate() and Maps.transliterate address maps that way). Registering under the system code made every map whose code differs from its file name resolve to a Hash-default empty entry — stages[:main].call crashed. 3106 of the sweep's 3175 failures were this crash. - Every alias and stage now carries doc_name. The Ruby/JS compilers emit run directives as Maps.transliterate(stage.doc_name, ...); without it, aliased-dependency runs (un-ell runs elot, 131 maps) compiled to transliterate(nil, s, :main). --- lib/interscript/isc/node_adapter.rb | 30 +++++++++++++++++------ spec/interscript/isc/node_adapter_spec.rb | 15 ++++++++++++ spec/isc_corpus_spec.rb | 10 ++++++++ 3 files changed, 48 insertions(+), 7 deletions(-) diff --git a/lib/interscript/isc/node_adapter.rb b/lib/interscript/isc/node_adapter.rb index 2110c940..04a52684 100644 --- a/lib/interscript/isc/node_adapter.rb +++ b/lib/interscript/isc/node_adapter.rb @@ -26,10 +26,22 @@ def build Interscript::Node::Document.new.tap do |doc| doc.metadata = build_metadata doc.tests = build_tests - doc.aliases = build_aliases - build_stages.each { |name, stage| doc.stages[name] = stage } + # Maps are addressed by their file-derived name (locate() and + # Maps.transliterate use it); registering under the system code + # made every map whose code differs from its file name resolve + # to an auto-created empty entry — stages[:main].call crashed. + doc.name = if @isc_doc[:filename] + File.basename(@isc_doc[:filename].to_s, ".isc") + else + @isc_doc[:systemCode] + end + # The legacy DSL stamps doc_name on every alias and stage; the + # Ruby/JS compilers emit run directives and alias lookups as + # Maps.transliterate(stage.doc_name, ...) — without it every + # aliased-dependency run compiled to transliterate(nil). + doc.aliases = build_aliases(doc.name) + build_stages(doc.name).each { |name, stage| doc.stages[name] = stage } build_dependencies(doc) - doc.name = @isc_doc[:systemCode] end end @@ -53,12 +65,14 @@ def build_tests tests end - def build_aliases + def build_aliases(doc_name) @isc_doc[:aliases].each_with_object({}) do |a, h| # The runtime resolves aliases through AliasDef#data (see # Interpreter::Stage#build_item); a bare item here hands it a # raw Array once Any#data unrolls. - h[a[:name].to_sym] = Interscript::Node::AliasDef.new(a[:name].to_sym, convert_item(a[:value])) + alias_def = Interscript::Node::AliasDef.new(a[:name].to_sym, convert_item(a[:value])) + alias_def.doc_name = doc_name + h[a[:name].to_sym] = alias_def end end @@ -90,9 +104,11 @@ def load_dependency_document(full_name) end end - def build_stages + def build_stages(doc_name) @isc_doc[:stages].each_with_object({}) do |stage, h| - h[stage[:name].to_sym] = build_stage(stage) + built = build_stage(stage) + built.doc_name = doc_name + h[stage[:name].to_sym] = built end end diff --git a/spec/interscript/isc/node_adapter_spec.rb b/spec/interscript/isc/node_adapter_spec.rb index 5941b2d1..a92944b6 100644 --- a/spec/interscript/isc/node_adapter_spec.rb +++ b/spec/interscript/isc/node_adapter_spec.rb @@ -358,3 +358,18 @@ def parse_and_adapt(src) expect(interp.call("αγκα")).to eq("αnκα") end end + +RSpec.describe "NodeAdapter document name derivation" do + it "names the document after the file, not the system code — transliteration addresses maps by file name" do + tree = Interscript::Isc::Parser.parse( + 'system "TOTALLY:diff-Code:frm:File" { stage main { sub { from "a" to "b" } } }' + ) + doc = Interscript::Isc::DocumentBuilder.build(tree, filename: "test-map.isc") + node = Interscript::Isc::NodeAdapter.to_interscript_node(doc) + # The Ruby compiler registers compiled maps under Document#name and + # Maps.transliterate looks them up by the map name (file-derived). + # Registering under the system code auto-creates an empty entry via + # the Hash default block and crashes on stages[:main].call. + expect(node.name).to eq("test-map") + end +end diff --git a/spec/isc_corpus_spec.rb b/spec/isc_corpus_spec.rb index 876125b0..e7434595 100644 --- a/spec/isc_corpus_spec.rb +++ b/spec/isc_corpus_spec.rb @@ -25,4 +25,14 @@ expect(Interscript.transliterate("bgnpcgn-deu-Latn-Latn-2000", "Tschüß!")) .to eq("Tschueß!") end + + it "compiles an aliased-dependency run through the Ruby compiler (un-ell runs elot)" do + skip "maps checkout not present" unless File.file?(File.expand_path("un-ell-Grek-Latn-1987-ts.isc", MAPS)) + + compiled = Interscript::Compiler::Ruby.call("un-ell-Grek-Latn-1987-ts") + # The run directive compiles to Maps.transliterate(stage.doc_name, ...); + # without doc_name stamped on ISC stages it emitted transliterate(nil) + # and crashed on stages[:main].call at runtime. + expect(compiled.call("Ένα")).to eq("Éna") + end end