From bbb9018b2b09a9f97517bfcf908816921e7bcfe0 Mon Sep 17 00:00:00 2001 From: Brian Searls Date: Mon, 27 Apr 2026 08:06:28 +0000 Subject: [PATCH 1/2] build: allow dead-code in interim tokenizer char class module --- src/v3/compiler/src/lib.rs | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/v3/compiler/src/lib.rs b/src/v3/compiler/src/lib.rs index 600f8eb538c..4f5396277a4 100644 --- a/src/v3/compiler/src/lib.rs +++ b/src/v3/compiler/src/lib.rs @@ -717,6 +717,8 @@ mod pipeline_authority; mod regen_parse_emit; mod regen_parse_tables_emit; mod tokenize; +#[allow(dead_code)] +mod tokenize_char_class; pub use regen_parse_emit::{render_parse_generated_rs, RenderParseGeneratedError}; pub use regen_parse_tables_emit::{ From 3408ea1f8b7c37c641335d246d2afca344493a57 Mon Sep 17 00:00:00 2001 From: Brian Searls Date: Mon, 27 Apr 2026 12:51:41 +0000 Subject: [PATCH 2/2] Fix tokenizer charclass witness and result/infer exactness checks --- src/v3/compiler/src/emit.rs | 280 +++++++++++++++++++++ src/v3/compiler/src/infer.rs | 67 ++++- src/v3/compiler/src/tokenize_char_class.rs | 149 +++++++++++ 3 files changed, 494 insertions(+), 2 deletions(-) create mode 100644 src/v3/compiler/src/tokenize_char_class.rs diff --git a/src/v3/compiler/src/emit.rs b/src/v3/compiler/src/emit.rs index 1edc68f1bb8..a2b14c151e8 100644 --- a/src/v3/compiler/src/emit.rs +++ b/src/v3/compiler/src/emit.rs @@ -63,6 +63,286 @@ use crate::variant_payload::{ }; use crate::Dag; +/// Result port of a finished `behavior` subgraph — shared by Go and Rust emit +/// paths for [`port_is_consumed_from`] (including `Behavior::Loop` body walks). +pub(super) fn behavior_result_port(behavior: &Behavior) -> PortId { + match behavior { + Behavior::Value(v) => v.result_port(), + Behavior::Transform(t) => t.result_port(), + Behavior::Branch(b) => b.result_port(), + Behavior::Loop(l) => l.result_port(), + Behavior::Bind(b) => b.result_port(), + } +} + +/// `std.error_primitives` declares canonical `Result = Ok { value: ok } | Err { value: err }`. +/// Emitters lower it to a target-native `Result` / `struct { Ok; Err }` carrier and must +/// not also emit a second substrate `type Result`. +/// +/// Suppression keys off the **resolved structural fingerprint** of that declaration +/// (name + type-parameter identities + `Ok`/`Err` payload wiring), not `span.file` +/// suffixes — so unrelated modules named `errors.dag` cannot collide, and renaming +/// the std file alone does not silently retarget suppression. +/// +/// **Policy:** the fingerprint is **global**, not std-scoped: any other declaration +/// named `Result` that matches this exact shape is also suppressed (intentional — the +/// substrate owns one canonical `Result` carrier; a user-defined twin with +/// the same fingerprint would not emit as a separate `type Result`). +pub(crate) fn substrate_result_type_decl_suppressed_for_emit( + dag: &Dag, + decl: &Declaration, +) -> bool { + if decl.name.as_deref() != Some("Result") { + return false; + } + let [ok_param, err_param] = match decl.type_params.as_slice() { + [a, b] => [*a, *b], + _ => return false, + }; + let ok_decl = dag.declaration(ok_param); + let err_decl = dag.declaration(err_param); + let ok_param_ok = matches!( + &ok_decl.connective, + TypeConnective::Atom(AtomPayload::TypeParam(name)) if name == "ok" + ); + let err_param_ok = matches!( + &err_decl.connective, + TypeConnective::Atom(AtomPayload::TypeParam(name)) if name == "err" + ); + if !ok_param_ok || !err_param_ok { + return false; + } + let TypeConnective::Disj { variants } = &decl.connective else { + return false; + }; + if variants.len() != 2 { + return false; + } + let Some(ok_field) = variants.iter().find(|v| v.label == "Ok") else { + return false; + }; + let Some(err_field) = variants.iter().find(|v| v.label == "Err") else { + return false; + }; + if variants.iter().filter(|v| v.label == "Ok").count() != 1 { + return false; + }; + if variants.iter().filter(|v| v.label == "Err").count() != 1 { + return false; + } + if variants.iter().any(|field| field.label != "Ok" && field.label != "Err") { + return false; + } + substrate_result_variant_payload_is_value_of(dag, ok_field.ty, ok_param) + && substrate_result_variant_payload_is_value_of(dag, err_field.ty, err_param) +} + +fn substrate_result_variant_payload_is_value_of( + dag: &Dag, + payload_ty: DeclarationId, + type_param: DeclarationId, +) -> bool { + let payload = dag.declaration(payload_ty); + let TypeConnective::Conj { children } = &payload.connective else { + return false; + }; + children.len() == 1 && children[0].label == "value" && children[0].ty == type_param +} + +pub(crate) fn dag_uses_arithmetic_div(dag: &Dag) -> bool { + dag.nodes().iter().any(|behavior| { + matches!( + behavior, + Behavior::Transform(TransformNode { + target: TransformTarget::Operator(OperatorKind::Arithmetic(ArithmeticOp::Div)), + .. + }) + ) + }) +} + +/// Structural port-liveness walk. Returns true if `target` appears as any port +/// reachable from `root` via producer→input edges (each behavior visits its +/// inputs once; cost is bounded by the body subgraph). +/// +/// Go and Rust emit use this to answer whether an arm body actually consumes a +/// payload binding — **without** scanning rendered source. The contract is a +/// graph fact, so the check stays structural. +/// +/// Ports with no producer (`produced_by = None`, as payload bindings use) are +/// leaves: the walk either hits `target` there or skips an unrelated parameter +/// port. +pub(super) fn port_is_consumed_from(dag: &Dag, root: PortId, target: PortId) -> bool { + if root == target { + return true; + } + let mut visited: HashSet = HashSet::new(); + let mut queue: Vec = vec![root]; + while let Some(port) = queue.pop() { + if !visited.insert(port) { + continue; + } + if port == target { + return true; + } + let Some(producer) = dag.port(port).produced_by else { + continue; + }; + match dag.node(producer) { + Behavior::Value(_) => {} + Behavior::Transform(t) => { + for input in t.inputs.iter().copied() { + queue.push(input); + } + } + Behavior::Branch(b) => { + queue.push(b.input); + for path in &b.paths { + queue.push(path.output); + } + } + Behavior::Loop(l) => { + queue.push(l.source); + queue.push(l.init); + if let Some(count) = l.bound.count_port() { + queue.push(count); + } + queue.push(behavior_result_port(dag.node(l.body))); + } + Behavior::Bind(b) => { + queue.push(b.value); + } + } + } + false +} + +/// Shared lookup failures for the emitter-internal type/operator walk helpers. +/// +/// **Dissolution note — 🟢 TERMINAL (local helper error coproduct).** Two +/// structurally distinct failure modes remain after consolidating the target +/// walkers: either the queried port has no resolved type yet, or the walk +/// reached a target-unsupported / malformed anchor and carries a diagnostic +/// string. The target emitters immediately map this local coproduct into their +/// own public error surfaces, so this enum is an implementation-layer bridge, +/// not a second user-facing authority. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(super) enum SharedEmitLookupError { + UntypedPort(PortId), + Unsupported(String), +} + +pub(super) fn primitive_type_id_for_port_shared( + dag: &Dag, + port: PortId, +) -> Result { + let ts = dag + .port(port) + .value_type() + .ok_or(SharedEmitLookupError::UntypedPort(port))?; + let mut current = ts.declaration; + for _ in 0..32 { + let decl = dag.declaration(current); + if decl.name.is_some() { + return Ok(current); + } + match &decl.connective { + TypeConnective::Instantiation { template, .. } => current = *template, + TypeConnective::Atom(AtomPayload::ResolvedByStructure(next)) + | TypeConnective::Atom(AtomPayload::ResolvedByName(next)) => current = *next, + _ => return Ok(current), + } + } + Err(SharedEmitLookupError::Unsupported( + "port type walk exceeded depth 32 — likely a cycle".to_string(), + )) +} + +pub(super) fn walk_to_disj(dag: &Dag, start: DeclarationId) -> Option { + let mut current = start; + for _ in 0..32 { + match &dag.declaration(current).connective { + TypeConnective::Disj { .. } => return Some(current), + TypeConnective::Cardinality { + bound: CardinalityBound::AtMostOne, + .. + } => return dag.optional_match_disj(current), + TypeConnective::Instantiation { template, .. } => current = *template, + TypeConnective::Atom(AtomPayload::ResolvedByStructure(next)) + | TypeConnective::Atom(AtomPayload::ResolvedByName(next)) => current = *next, + _ => return None, + } + } + None +} + +pub(super) fn algebra_field_for_operator_shared( + dag: &Dag, + operand_type_id: DeclarationId, + op: OperatorKind, +) -> Result { + let Some(algebra_conj_id) = walk_to_algebra_conj(dag, operand_type_id) else { + return canonical_operator_field_shared(dag, op); + }; + let field_label = crate::operators::algebra_field_name(op); + let children = match &dag.declaration(algebra_conj_id).connective { + TypeConnective::Conj { children } => children, + _ => unreachable!("walk_to_algebra_conj returned a non-Conj"), + }; + if let Some(field) = children.iter().find(|field| field.label == field_label) { + return Ok(field.ty); + } + canonical_operator_field_shared(dag, op) +} + +fn walk_to_algebra_conj(dag: &Dag, start: DeclarationId) -> Option { + let mut current = start; + for _ in 0..32 { + let decl = dag.declaration(current); + match &decl.connective { + TypeConnective::Conj { .. } => return Some(current), + TypeConnective::Instantiation { template, .. } => current = *template, + TypeConnective::Atom(AtomPayload::ResolvedByStructure(next)) + | TypeConnective::Atom(AtomPayload::ResolvedByName(next)) => current = *next, + _ => { + if let Some(inh) = decl.inhabits { + current = inh; + } else { + return None; + } + } + } + } + None +} + +fn canonical_operator_field_shared( + dag: &Dag, + op: OperatorKind, +) -> Result { + let ordered_ring_id = dag.ordered_ring_decl().ok_or_else(|| { + SharedEmitLookupError::Unsupported( + "bootstrap is missing the canonical `OrderedRing` declaration".to_string(), + ) + })?; + let ordered_ring = dag.declaration(ordered_ring_id); + let TypeConnective::Conj { children } = &ordered_ring.connective else { + return Err(SharedEmitLookupError::Unsupported( + "`OrderedRing` does not lower to a Conj declaration".to_string(), + )); + }; + let field_label = crate::operators::algebra_field_name(op); + children + .iter() + .find(|field| field.label == field_label) + .map(|field| field.ty) + .ok_or_else(|| { + SharedEmitLookupError::Unsupported(format!( + "`OrderedRing` has no canonical field labeled {field_label}" + )) + }) +} + #[derive(Debug, Clone, PartialEq, Eq)] enum GoFieldAccessBinding { DirectField(String), diff --git a/src/v3/compiler/src/infer.rs b/src/v3/compiler/src/infer.rs index df6e35f2ac9..71302876a97 100644 --- a/src/v3/compiler/src/infer.rs +++ b/src/v3/compiler/src/infer.rs @@ -3874,8 +3874,71 @@ fn substitute_receiver( | TypeConnective::Atom(AtomPayload::ResolvedByName(next)) => { substitute_receiver(dag, *next, receiver_param, source_id) } - TypeConnective::Instantiation { template, .. } => { - substitute_receiver(dag, *template, receiver_param, source_id) + TypeConnective::Instantiation { + template, + arguments, + } => { + let mut new_args: Vec = Vec::with_capacity(arguments.len()); + let mut any_change = false; + let mut any_unresolved = false; + for arg in arguments { + let Some(new_val) = substitute_receiver(dag, arg.value, receiver_param, source_id) else { + any_unresolved = true; + new_args.push(*arg); + continue; + }; + if new_val != arg.value { + any_change = true; + } + new_args.push(TemplateArgument { + parameter: arg.parameter, + value: new_val, + }); + } + if any_unresolved { + return if any_change { None } else { Some(current) }; + } + if !any_change { + return Some(current); + } + // Same dedup contract as `div_total_result_output_shape`: false negatives + // here allocate unbounded anonymous instantiations across fixpoint iterations. + if let Some(existing) = find_equivalent_anonymous_instantiation( + dag, + *template, + &new_args, + decl.nominal_opacity.as_ref(), + ) { + return Some(existing); + } + let id = dag.alloc_declaration_id(); + let span = decl.span.clone(); + dag.push_declaration(Declaration { + id, + name: None, + connective: TypeConnective::Instantiation { + template: *template, + arguments: new_args, + }, + type_params: Vec::new(), + phantom_params: Vec::new(), + meta_tag: None, + specialization_parent: None, + inhabits: None, + value_body: None, + refinement: None, + nominal_opacity: decl.nominal_opacity.clone(), + span, + }); + Some(id) + } + _ => { + if decl.name.is_some() { + Some(current) + } else { + None + } + } } // Non-receiver TypeParam in an algebra field (e.g., a // second generic parameter) isn't resolvable at M1(2.7). diff --git a/src/v3/compiler/src/tokenize_char_class.rs b/src/v3/compiler/src/tokenize_char_class.rs new file mode 100644 index 00000000000..b1871bce2ac --- /dev/null +++ b/src/v3/compiler/src/tokenize_char_class.rs @@ -0,0 +1,149 @@ +//! ASCII-oriented lexical classes for the SG-1 tokenizer. +//! +//! Authority: `dsl/std/unicode.dag` — `CharClass` + `char_in_class` define the +//! same ASCII scalar predicates. This module is the Rust projection the +//! `regen_tokenize` binary emits into `tokenize_generated.rs` because M1(2.8) +//! still treats list / sum-variant `data` bodies in `tokenize.dag` as opaque +//! (`DOWNSTREAM_REQUIREMENTS.md` class-5 gap #3), so the scanner cannot yet +//! read `CharClass` rows structurally from that authority file. +//! +//! `byte_matches` is **hand-synced** with `char_in_class` in `dsl/std/unicode.dag` +//! on code points U+0000–U+007F (same Int-range semantics). There is no runtime +//! bridge from lowered `.dag` yet: `#[cfg(test)]` checks lock the mirror against +//! Rust’s `u8::is_ascii_*` and use substring anchors on `unicode.dag` source — not +//! evaluated `char_in_class`. That is intentional **bounded debt** under +//! modeling-discipline P2 (single authority / drift risk): the executable truth +//! for predicates is only in `.dag` today. **ROADMAP** row *`char_in_class` +//! interpreter parity (tokenizer bridge finish, PR #693)* tracks replacing the +//! host-predicate ratchet with `0..=127` parity vs evaluated `char_in_class` once +//! the v3 test harness can run it (same gate as deleting this mirror). +//! +//! **Lane framing:** tokenizer-side interim only — not structural consumption of +//! `CharClass` from lowered `tokenize.dag` (see M1(2.8) class-5 gap #3). Remove +//! this module **and** [`TokenizerCharClass`] when `regen_tokenize` reads class +//! predicates from lowered `.dag` — do not leave a duplicate Rust sum of variant +//! names beside the `.dag` authority. + +/// Mirrors `std.unicode::CharClass` variant names for `tokenize_generated.rs` call +/// sites only. **Dissolution:** delete this enum with this module when structural +/// scanner rows land — not just the `byte_matches` body. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum TokenizerCharClass { + Whitespace, + Digit, + IdentStart, + IdentContinue, +} + +#[inline] +pub(crate) fn byte_matches(byte: u8, class: TokenizerCharClass) -> bool { + let cp = byte as i64; + match class { + // Match Rust `u8::is_ascii_whitespace` (excludes vertical tab U+000B). + TokenizerCharClass::Whitespace => { + matches!(byte, b'\t' | b'\n' | b'\x0C' | b'\r' | b' ') + } + TokenizerCharClass::Digit => (48..=57).contains(&cp), + TokenizerCharClass::IdentStart => { + (65..=90).contains(&cp) || (97..=122).contains(&cp) || cp == 95 + } + // Match `unicode.dag` `IdentContinue` arm shape: inline digit range (same + // as `Digit`), then letter/underscore — no `byte_matches` self-call so + // the mirror stays line-aligned with the `.dag` CX choice. + TokenizerCharClass::IdentContinue => { + (48..=57).contains(&cp) + || (65..=90).contains(&cp) + || (97..=122).contains(&cp) + || cp == 95 + } + } +} + +#[cfg(test)] +mod sub_charclass_in_std_unicode_gate { + //! **Layer:** unit (TESTING.md) — T-Sub `sub_charclass_in_std_unicode` tokenizer + //! half / bounded interim (see ROADMAP.md: T-Sub lane + `char_in_class` interpreter + //! parity row). Ratchets `std.unicode` + generated tokenizer wiring + `byte_matches` + //! vs `u8::is_ascii_*` on 0..=127 (behavioral until interpreter parity lands). + //! + //! **Sync boundary:** parity here is `byte_matches` vs `u8::is_ascii_*`, not + //! evaluated `char_in_class` from `.dag`; keep `tokenize_char_class.rs` and + //! `unicode.dag` predicates edited together until an interpreter-backed check exists. + //! Substring anchors below only catch gross drift (missing sum, restored host + //! `is_ascii_*`, accidental reintroduction of `char_in_class` self-call on the + //! same `c` in `IdentContinue`); they do not prove arithmetic matches `.dag` and + //! are fragile to harmless `unicode.dag` reformatting. The `0..=127` + //! `byte_matches` vs `is_ascii_*` loop is the primary behavioral guard until + //! interpreter-backed parity (ROADMAP) replaces these anchors. + + use super::{byte_matches, TokenizerCharClass}; + + const UNICODE_DAG: &str = include_str!("../../../../dsl/std/unicode.dag"); + const TOKENIZE_GENERATED: &str = include_str!("tokenize_generated.rs"); + + #[test] + fn unicode_dag_defines_char_class() { + assert!( + UNICODE_DAG.contains("type CharClass") + && UNICODE_DAG.contains("fn char_in_class") + && UNICODE_DAG.contains("Whitespace") + && UNICODE_DAG.contains("IdentContinue"), + "expected `CharClass` sum + `char_in_class` predicate in `dsl/std/unicode.dag` authority" + ); + assert!( + UNICODE_DAG.contains("IdentContinue => (cp >= 48 && cp <= 57)"), + "expected `IdentContinue` to inline the digit range (no `char_in_class` recursion on the same `c`)" + ); + assert!( + !UNICODE_DAG.contains("char_in_class(c: c, class: Digit)"), + "`IdentContinue` must not recurse into `char_in_class` with the same `c` (CX / bounded recursion)" + ); + } + + #[test] + fn generated_tokenizer_avoids_ascii_host_predicates() { + assert!( + !TOKENIZE_GENERATED.contains("is_ascii_whitespace") + && !TOKENIZE_GENERATED.contains("is_ascii_digit") + && !TOKENIZE_GENERATED.contains("is_ascii_alphabetic") + && !TOKENIZE_GENERATED.contains("is_ascii_alphanumeric"), + "tokenize_generated.rs should route ASCII classes through `byte_matches` / `TokenizerCharClass`, \ + not std-lib `is_ascii_*` helpers" + ); + assert!( + TOKENIZE_GENERATED.contains("byte_matches") + && TOKENIZE_GENERATED.contains("ScannerCharClass::Whitespace") + && TOKENIZE_GENERATED.contains("ScannerCharClass::Digit") + && TOKENIZE_GENERATED.contains("ScannerCharClass::IdentStart") + && TOKENIZE_GENERATED.contains("ScannerCharClass::IdentContinue"), + "expected structural CharClass projection wired into generated tokenizer" + ); + } + + #[test] + fn byte_matches_locks_ascii_scanner_semantics() { + for byte in 0u8..=127 { + let b = byte; + assert_eq!( + byte_matches(b, TokenizerCharClass::Whitespace), + b.is_ascii_whitespace(), + "Whitespace mismatch at {byte}" + ); + assert_eq!( + byte_matches(b, TokenizerCharClass::Digit), + b.is_ascii_digit(), + "Digit mismatch at {byte}" + ); + assert_eq!( + byte_matches(b, TokenizerCharClass::IdentStart), + b.is_ascii_alphabetic() || b == b'_', + "IdentStart mismatch at {byte}" + ); + assert_eq!( + byte_matches(b, TokenizerCharClass::IdentContinue), + b.is_ascii_alphanumeric() || b == b'_', + "IdentContinue mismatch at {byte}" + ); + } + } +}