diff --git a/dag/gunbc/char_at_scaling_probe_support.dag b/dag/gunbc/char_at_scaling_probe_support.dag new file mode 100644 index 00000000000..4059433e695 --- /dev/null +++ b/dag/gunbc/char_at_scaling_probe_support.dag @@ -0,0 +1,28 @@ +module gunbc.char_at_scaling_probe_support + +import std.types { NonEmptyStr } +import std.disposition { Disposition, Terminal } +import std.dissolution { DissolutionCondition, unbound_dissolution } + +// Per DESIGN §4b: this probe IS the discriminating evidence for CHARAT-0's O(1) +// char_at claim, so it stays enrolled rather than dissolving — the disposition is +// Terminal, not Scaffold. §5's same-unit rule is why this carries its own +// independent trigger rather than sharing json_parse_scaling_probe.rs's: the two +// probes measure different primitives (char_at cost in isolation vs. end-to-end +// parse) and could have independently different lifetimes. +data char_at_scaling_probe_disposition: Disposition = Terminal { + reason: "the entry point `src/v1/stage0/src/bin/char_at_scaling_probe.rs` resolves against — a single `char_at` call, isolated from every other primitive, so per-access cost can be fit against position and against string length without the JSON parser's independent list-materialization cost. NOT floor-enrolled; the acceptance evidence for CHARAT-0's O(1) claim." +} + +// §3: cite the symbol, not a position or another file's marker. The bin's own +// grep receipt is `CHAR_AT_SCALING_PROBE_SCAFFOLD_MARKER` in +// `char_at_scaling_probe.rs` — a real, distinct const from +// `json_parse_scaling_probe.rs`'s `JSON_PARSE_SCALING_PROBE_SCAFFOLD_MARKER`, +// which this probe's disposition no longer cites or depends on. +data char_at_scaling_probe_dissolution: DissolutionCondition = unbound_dissolution( + description: "delete `src/v1/stage0/src/bin/char_at_scaling_probe.rs` (grep receipt: `CHAR_AT_SCALING_PROBE_SCAFFOLD_MARKER`) when char_at's O(1) property is floor-enrolled with a modeled witness, or when a fresh run of this probe's own printed TSV (mean_call_us per string_len/position, per its CHECKABLE RECEIPT doc comment) shows mean_call_us growing with string_len across the CHAR_AT_PROBE_LENGTHS range instead of staying flat" as NonEmptyStr +) + +fn char_at_probe(s: String, pos: Int) -> String { + char_at(s: s, pos: pos) +} diff --git a/src/v1/runtime_rust.dag b/src/v1/runtime_rust.dag index de9927b7454..247c8ca1969 100644 --- a/src/v1/runtime_rust.dag +++ b/src/v1/runtime_rust.dag @@ -95,23 +95,34 @@ fn rt_concat_trait() -> String { fn rt_string_ops() -> String { concat( - "pub fn char_at(s: &str, pos: i64) -> String {\n", + "/// Ascii-aware variant taking a precomputed `is_ascii` flag (the `RcStr` carrier fact)\n", + "/// instead of rescanning the whole string on every call -- the per-call `s.is_ascii()`\n", + "/// scan is what made repeated indexing over a large string O(n^2) (STRING-INDEX-0).\n", + "pub fn char_at_ascii_aware(s: &str, is_ascii: bool, pos: i64) -> String {\n", " let pos = pos.max(0) as usize;\n", - " if s.is_ascii() {\n", + " if is_ascii {\n", " let bytes = s.as_bytes();\n", " if pos >= bytes.len() { return String::new(); }\n", " return String::from(bytes[pos] as char);\n", " }\n", " s.chars().nth(pos).map(|ch| ch.to_string()).unwrap_or_default()\n", "}\n\n", + "pub fn char_at(s: &str, pos: i64) -> String {\n", + " char_at_ascii_aware(s, s.is_ascii(), pos)\n", + "}\n\n", + "/// Ascii-aware variant taking a precomputed `is_ascii` flag; see `char_at_ascii_aware`.\n", + "pub fn string_length_ascii_aware(s: &str, is_ascii: bool) -> i64 {\n", + " if is_ascii { s.len() as i64 } else { s.chars().count() as i64 }\n", + "}\n\n", "pub fn string_length(s: &str) -> i64 {\n", - " if s.is_ascii() { s.len() as i64 } else { s.chars().count() as i64 }\n", + " string_length_ascii_aware(s, s.is_ascii())\n", "}\n\n", - "pub fn substring(s: &str, start: i64, end: i64) -> String {\n", + "/// Ascii-aware variant taking a precomputed `is_ascii` flag; see `char_at_ascii_aware`.\n", + "pub fn substring_ascii_aware(s: &str, is_ascii: bool, start: i64, end: i64) -> String {\n", " let start = start.max(0) as usize;\n", " let end = end.max(0) as usize;\n", " if end <= start { return String::new(); }\n", - " if s.is_ascii() {\n", + " if is_ascii {\n", " let len = s.len();\n", " if start >= len { return String::new(); }\n", " let out_end = end.min(len);\n", @@ -123,6 +134,9 @@ fn rt_string_ops() -> String { " record_substring_chars_walked(s, start, take_len);\n", " s.chars().skip(start).take(take_len).collect()\n", "}\n\n", + "pub fn substring(s: &str, start: i64, end: i64) -> String {\n", + " substring_ascii_aware(s, s.is_ascii(), start, end)\n", + "}\n\n", "pub fn string_contains(s: &str, sub: String) -> bool {\n", " s.contains(&*sub)\n", "}\n\n", diff --git a/src/v1/stage0/Cargo.toml b/src/v1/stage0/Cargo.toml index 81339a292ce..a2dbc2fe461 100644 --- a/src/v1/stage0/Cargo.toml +++ b/src/v1/stage0/Cargo.toml @@ -89,6 +89,13 @@ path = "src/bin/claim_executor.rs" name = "json_parse_scaling_probe" path = "src/bin/json_parse_scaling_probe.rs" +# Isolated char_at scaling receipt: position x string-length, uncontaminated by the +# JSON parser's independent list-accumulation cost (eager-koi-458 scope correction, +# 2026-08-17). DISSOLUTION: delete alongside json_parse_scaling_probe.rs. +[[bin]] +name = "char_at_scaling_probe" +path = "src/bin/char_at_scaling_probe.rs" + # Resolved-type owned-data discovery for Consolidation #4553 glob discovery. # Hand-written CI tool (like claim_batch); exposes neutral decl facts only. [[bin]] diff --git a/src/v1/stage0/src/bin/char_at_scaling_probe.rs b/src/v1/stage0/src/bin/char_at_scaling_probe.rs new file mode 100644 index 00000000000..dfe234681d0 --- /dev/null +++ b/src/v1/stage0/src/bin/char_at_scaling_probe.rs @@ -0,0 +1,142 @@ +#![allow(clippy::disallowed_macros)] + +//! SCAFFOLD (DESIGN §7 seed-retained HAND-RUST / CHARAT-0) — isolated `char_at` scaling +//! receipt: many repeated calls at varying position and varying string length, uncontaminated +//! by the JSON parser's independent `value_to_list_carrier` / `free_monoid_to_vec` +//! materialization cost (a different, separately scoped defect — eager-koi-458's 2026-08-17 +//! scope correction on this work item names it and rules it out of this instrument). +//! +//! `json_parse_scaling_probe`'s end-to-end parse measures whichever quadratic dominates and +//! cannot by itself distinguish "char_at is O(1)" from "the list carrier is O(m^2)". This bin +//! isolates the one thing STRING-INDEX-0 changed: `char_at` no longer rescans `is_ascii` per +//! call, because ascii-ness is now a precomputed fact carried on `RcStr`. On an ASCII input, +//! per-call cost should therefore be flat against BOTH position and string length. +//! +//! NOT floor-enrolled — run standalone via `cargo run --release -p v1-compiler --bin +//! char_at_scaling_probe`. Entry: `dag/gunbc/char_at_scaling_probe_support.dag`'s +//! `char_at_probe(s, pos)`, a one-line wrapper around the `char_at` free call. +//! +//! CHECKABLE RECEIPT: for each (length, position) pair, mean per-call elapsed time over +//! `CHAR_AT_PROBE_REPS` repeated calls through the interpreter — printed as TSV. +//! +//! DISSOLUTION (own trigger, independent of `json_parse_scaling_probe.rs`'s — see +//! `char_at_scaling_probe_dissolution` in `dag/gunbc/char_at_scaling_probe_support.dag`, +//! DESIGN §5's same-unit rule): delete this bin when CHARAT-0's `char_at` O(1) property is +//! floor-enrolled with a modeled witness, or when a fresh run's own printed TSV (the +//! CHECKABLE RECEIPT below) shows `mean_call_us` growing with `string_len` across the +//! `CHAR_AT_PROBE_LENGTHS` range instead of staying flat. +//! Receipt: `rg CHAR_AT_SCALING_PROBE_SCAFFOLD_MARKER src/v1/stage0` until deletion. + +/// Grep receipt for scaffold dissolution (`rg CHAR_AT_SCALING_PROBE_SCAFFOLD_MARKER`). +pub const CHAR_AT_SCALING_PROBE_SCAFFOLD_MARKER: &str = + "CHARAT-0 char_at_scaling_probe measurement transport (not floor-enrolled)"; + +use std::process::ExitCode; +use std::time::Instant; + +use v1_compiler::cli_run::{make_eval_context, resolve_entry_graph, workspace_root}; +use v1_compiler::v1_interpreter::{self, str_value, ExecutionMode, Value}; + +const ENTRY: &str = "dag/gunbc/char_at_scaling_probe_support.dag"; + +/// One repeated call per (length, position) pair; small enough to keep the whole +/// grid under a minute, large enough to average out interpreter dispatch noise. +const DEFAULT_REPS: usize = 2000; + +fn reps_from_env() -> usize { + std::env::var("CHAR_AT_PROBE_REPS") + .ok() + .and_then(|s| s.parse().ok()) + .unwrap_or(DEFAULT_REPS) +} + +/// Pure-ASCII fill so the fast path (byte-offset indexing, no `.chars().nth` walk) is +/// what's being timed — this is the branch STRING-INDEX-0 made O(1). +fn make_ascii_string(len: usize) -> String { + (0..len).map(|i| (b'a' + (i % 26) as u8) as char).collect() +} + +fn resolve_ctx() -> Result { + let ws = workspace_root(); + let roots = vec![ws.join("dag").to_string_lossy().into_owned()]; + eprintln!("char_at_scaling_probe: resolving {ENTRY} ..."); + let resolve_start = Instant::now(); + let (graph, indices) = resolve_entry_graph(&roots, ENTRY)?; + eprintln!( + "char_at_scaling_probe: resolve_ms={}", + resolve_start.elapsed().as_millis() + ); + Ok(make_eval_context(&graph, indices, ExecutionMode::Hermetic)) +} + +fn call_char_at_once( + ctx: &v1_interpreter::InterpContext, + s: &Value, + pos: i64, +) -> Result<(), String> { + let args = [ + (Some("s".to_string()), s.clone()), + (Some("pos".to_string()), Value::Int(pos)), + ]; + v1_interpreter::run_in_context_with_args(ctx, "char_at_probe", &args, false) + .map(|_| ()) + .map_err(|e| format!("char_at_probe: {e}")) +} + +/// Mean per-call elapsed microseconds over `reps` calls at one fixed (s, pos). +fn mean_call_us( + ctx: &v1_interpreter::InterpContext, + s: &Value, + pos: i64, + reps: usize, +) -> Result { + // One untimed warmup call so any first-call setup (e.g. lazy memo frame init) + // doesn't bias the measured mean. + call_char_at_once(ctx, s, pos)?; + let start = Instant::now(); + for _ in 0..reps { + call_char_at_once(ctx, s, pos)?; + } + let elapsed = start.elapsed(); + Ok(elapsed.as_secs_f64() * 1_000_000.0 / reps as f64) +} + +fn run() -> Result<(), String> { + let ctx = resolve_ctx()?; + let reps = reps_from_env(); + let lengths: Vec = std::env::var("CHAR_AT_PROBE_LENGTHS") + .ok() + .map(|s| { + s.split(',') + .filter_map(|tok| tok.trim().parse().ok()) + .collect() + }) + .filter(|v: &Vec| !v.is_empty()) + .unwrap_or_else(|| vec![10_000, 100_000, 500_000, 2_000_000]); + let position_fractions: [f64; 5] = [0.0, 0.25, 0.5, 0.75, 0.99]; + + eprintln!("char_at_scaling_probe: reps={reps} lengths={lengths:?}"); + println!("mode\tchar_at_scaling\treps={reps}"); + println!("string_len\tposition\tposition_frac\tmean_call_us"); + + for &len in &lengths { + let s = make_ascii_string(len); + let value = str_value(&s); + for frac in position_fractions { + let pos = ((len.saturating_sub(1)) as f64 * frac).round() as i64; + let mean_us = mean_call_us(&ctx, &value, pos, reps)?; + println!("{len}\t{pos}\t{frac:.2}\t{mean_us:.3}"); + } + } + Ok(()) +} + +fn main() -> ExitCode { + match run() { + Ok(()) => ExitCode::SUCCESS, + Err(e) => { + eprintln!("char_at_scaling_probe: ERROR: {e}"); + ExitCode::FAILURE + } + } +} diff --git a/src/v1/stage0/src/bin/json_parse_scaling_probe.rs b/src/v1/stage0/src/bin/json_parse_scaling_probe.rs index 13c71b0ecb5..0224ae229c1 100644 --- a/src/v1/stage0/src/bin/json_parse_scaling_probe.rs +++ b/src/v1/stage0/src/bin/json_parse_scaling_probe.rs @@ -7,6 +7,8 @@ //! json_parse_scaling_probe`. Modes (`JSON_PARSE_PROBE_MODE`): //! - `survival`: one `JSON_PARSE_TARGET_BYTES`, exactly one `parse_json` call (fresh process). //! - `memo_receipt`: one target; cold parse + first repeat + average of subsequent memo hits. +//! - `counters`: one target; reports `MutationCounters` (list_push calls vs items_copied) to +//! distinguish an O(log n) push_back from an O(m) `value_to_list_carrier` fallback copy. //! - `scaling` / `large`: legacy grids (memo-contaminated; not used for acceptance receipts). //! //! CHECKABLE RECEIPT: survival mode records Present parse + member count vs process death. @@ -24,11 +26,7 @@ use std::process::ExitCode; use std::time::Instant; use v1_compiler::cli_run::{make_eval_context, resolve_entry_graph, workspace_root}; -use v1_compiler::v1_interpreter::{self, ExecutionMode, Value}; - -fn str_value(s: impl AsRef) -> Value { - Value::Str(std::rc::Rc::from(s.as_ref())) -} +use v1_compiler::v1_interpreter::{self, str_value, ExecutionMode, Value}; const ENTRY: &str = "dag/extdeps/languages/json/parse.dag"; @@ -204,6 +202,39 @@ fn survival_succeeded(outcome: &ParseOutcome) -> bool { matches!(outcome, ParseOutcome::Parsed { .. }) } +/// One target, one parse — reports `MutationCounters` (list_push calls vs items copied) +/// to distinguish an O(log n) push_back from an O(m) `value_to_list_carrier` fallback +/// materialization. Cheap/decisive per eager-koi-458's 2026-08-17 scope correction. +fn run_counters(ctx: &v1_interpreter::InterpContext) -> Result { + let target = target_bytes_from_env()?; + let (member_count, json) = json_object_at_least_bytes(target); + eprintln!( + "json_parse_scaling_probe: counters target={target} members={member_count} bytes={}", + json.len() + ); + let before = ctx.mutation_counters_snapshot(); + let outcome = parse_json_once(ctx, &json, member_count); + let after = ctx.mutation_counters_snapshot(); + println!("mode\tcounters\toutcome={}", outcome_label(&outcome)); + println!("counter\tcalls\titems_copied"); + println!( + "list_push\t{}\t{}", + after.list_push_calls - before.list_push_calls, + after.list_push_items_copied - before.list_push_items_copied + ); + println!( + "list_concat\t{}\t{}", + after.list_concat_calls - before.list_concat_calls, + after.list_concat_items_copied - before.list_concat_items_copied + ); + println!( + "map_insert\t{}\t{}", + after.map_insert_calls - before.map_insert_calls, + after.map_insert_entries_copied - before.map_insert_entries_copied + ); + Ok(outcome) +} + /// One target, one parse — intended for a fresh process per invocation. fn run_survival(ctx: &v1_interpreter::InterpContext) -> Result { let target = target_bytes_from_env()?; @@ -343,6 +374,7 @@ fn run() -> Result { let mode = std::env::var("JSON_PARSE_PROBE_MODE").unwrap_or_else(|_| "scaling".to_string()); let success = match mode.as_str() { "survival" => survival_succeeded(&run_survival(&ctx)?), + "counters" => survival_succeeded(&run_counters(&ctx)?), "memo_receipt" => { run_memo_receipt(&ctx)?; true diff --git a/src/v1/stage0/src/derived_realization_schedule.rs b/src/v1/stage0/src/derived_realization_schedule.rs index 1cd938f3dfd..0b432e2d794 100644 --- a/src/v1/stage0/src/derived_realization_schedule.rs +++ b/src/v1/stage0/src/derived_realization_schedule.rs @@ -92,7 +92,7 @@ pub fn realize_pack_width_from_scalars( _ => -1, }; let verdict = match realize_ctx.field(&fields, "verdict") { - Some(Value::Str(s)) => Rc::clone(s), + Some(Value::Str(s)) => s.rc(), _ => Rc::from("unknown"), }; Ok(DerivedScheduleWidth { diff --git a/src/v1/stage0/src/recorded_fixture.rs b/src/v1/stage0/src/recorded_fixture.rs index 20b426e7b2f..83922bd1873 100644 --- a/src/v1/stage0/src/recorded_fixture.rs +++ b/src/v1/stage0/src/recorded_fixture.rs @@ -413,7 +413,7 @@ pub fn value_to_fixture_json( Value::Bool(b) => Ok(json!({ "__tag": "Bool", "value": b })), Value::Int(n) => Ok(json!({ "__tag": "Int", "value": n })), Value::Float(f) => Ok(json!({ "__tag": "Float", "value": f })), - Value::Str(s) => Ok(json!({ "__tag": "Str", "value": s })), + Value::Str(s) => Ok(json!({ "__tag": "Str", "value": s.as_ref() })), Value::List(items) => { let arr: Result, _> = items .iter() diff --git a/src/v1/stage0/src/v1_compiler_runtime_rust.rs b/src/v1/stage0/src/v1_compiler_runtime_rust.rs index e62b051f41f..8ba40354723 100644 --- a/src/v1/stage0/src/v1_compiler_runtime_rust.rs +++ b/src/v1/stage0/src/v1_compiler_runtime_rust.rs @@ -36,7 +36,7 @@ pub fn rt_concat_trait() -> String { } pub fn rt_string_ops() -> String { - v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat("pub fn char_at(s: &str, pos: i64) -> String {\n".to_string(), " let pos = pos.max(0) as usize;\n".to_string()), " if s.is_ascii() {\n".to_string()), " let bytes = s.as_bytes();\n".to_string()), " if pos >= bytes.len() { return String::new(); }\n".to_string()), " return String::from(bytes[pos] as char);\n".to_string()), " }\n".to_string()), " s.chars().nth(pos).map(|ch| ch.to_string()).unwrap_or_default()\n".to_string()), "}\n\n".to_string()), "pub fn string_length(s: &str) -> i64 {\n".to_string()), " if s.is_ascii() { s.len() as i64 } else { s.chars().count() as i64 }\n".to_string()), "}\n\n".to_string()), "pub fn substring(s: &str, start: i64, end: i64) -> String {\n".to_string()), " let start = start.max(0) as usize;\n".to_string()), " let end = end.max(0) as usize;\n".to_string()), " if end <= start { return String::new(); }\n".to_string()), " if s.is_ascii() {\n".to_string()), " let len = s.len();\n".to_string()), " if start >= len { return String::new(); }\n".to_string()), " let out_end = end.min(len);\n".to_string()), " let take_len = out_end.saturating_sub(start);\n".to_string()), " record_substring_chars_walked(s, start, take_len);\n".to_string()), " return s[start..out_end].to_string();\n".to_string()), " }\n".to_string()), " let take_len = end.saturating_sub(start);\n".to_string()), " record_substring_chars_walked(s, start, take_len);\n".to_string()), " s.chars().skip(start).take(take_len).collect()\n".to_string()), "}\n\n".to_string()), "pub fn string_contains(s: &str, sub: String) -> bool {\n".to_string()), " s.contains(&*sub)\n".to_string()), "}\n\n".to_string()), "pub fn contains(s: String, sub: String) -> bool { string_contains(&s, sub) }\n\n".to_string()), "pub fn starts_with(s: String, prefix: String) -> bool { s.starts_with(&*prefix) }\n\n".to_string()), "pub fn ends_with(s: String, suffix: String) -> bool { s.ends_with(&*suffix) }\n\n".to_string()), "pub fn trim(s: String) -> String { s.trim().to_string() }\n\n".to_string()), "pub fn count(items: Rc>) -> i64 { items.len() as i64 }\n\n".to_string()), "pub fn str_eq(a: String, b: String) -> bool { a == b }\n\n".to_string()), "pub fn to_string(value: i64) -> String { value.to_string() }\n\n".to_string()), "/// RFC 3629 UTF-8 decode of a byte vector. Fail-closed on invalid UTF-8.\n".to_string()), "pub fn utf8_decode_bytes(bytes: &[u8]) -> Result {\n".to_string()), " String::from_utf8(bytes.to_vec()).map_err(|e| format!(\"invalid UTF-8 in access payload: {}\", e))\n".to_string()), "}\n\n".to_string()), "pub fn clamp(val: i64, min_val: i64, max_val: i64) -> i64 {\n".to_string()), " val.clamp(min_val, max_val)\n".to_string()), "}\n\n".to_string()) + v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat(v1_rt::concat("/// Ascii-aware variant taking a precomputed `is_ascii` flag (the `RcStr` carrier fact)\n".to_string(), "/// instead of rescanning the whole string on every call -- the per-call `s.is_ascii()`\n".to_string()), "/// scan is what made repeated indexing over a large string O(n^2) (STRING-INDEX-0).\n".to_string()), "pub fn char_at_ascii_aware(s: &str, is_ascii: bool, pos: i64) -> String {\n".to_string()), " let pos = pos.max(0) as usize;\n".to_string()), " if is_ascii {\n".to_string()), " let bytes = s.as_bytes();\n".to_string()), " if pos >= bytes.len() { return String::new(); }\n".to_string()), " return String::from(bytes[pos] as char);\n".to_string()), " }\n".to_string()), " s.chars().nth(pos).map(|ch| ch.to_string()).unwrap_or_default()\n".to_string()), "}\n\n".to_string()), "pub fn char_at(s: &str, pos: i64) -> String {\n".to_string()), " char_at_ascii_aware(s, s.is_ascii(), pos)\n".to_string()), "}\n\n".to_string()), "/// Ascii-aware variant taking a precomputed `is_ascii` flag; see `char_at_ascii_aware`.\n".to_string()), "pub fn string_length_ascii_aware(s: &str, is_ascii: bool) -> i64 {\n".to_string()), " if is_ascii { s.len() as i64 } else { s.chars().count() as i64 }\n".to_string()), "}\n\n".to_string()), "pub fn string_length(s: &str) -> i64 {\n".to_string()), " string_length_ascii_aware(s, s.is_ascii())\n".to_string()), "}\n\n".to_string()), "/// Ascii-aware variant taking a precomputed `is_ascii` flag; see `char_at_ascii_aware`.\n".to_string()), "pub fn substring_ascii_aware(s: &str, is_ascii: bool, start: i64, end: i64) -> String {\n".to_string()), " let start = start.max(0) as usize;\n".to_string()), " let end = end.max(0) as usize;\n".to_string()), " if end <= start { return String::new(); }\n".to_string()), " if is_ascii {\n".to_string()), " let len = s.len();\n".to_string()), " if start >= len { return String::new(); }\n".to_string()), " let out_end = end.min(len);\n".to_string()), " let take_len = out_end.saturating_sub(start);\n".to_string()), " record_substring_chars_walked(s, start, take_len);\n".to_string()), " return s[start..out_end].to_string();\n".to_string()), " }\n".to_string()), " let take_len = end.saturating_sub(start);\n".to_string()), " record_substring_chars_walked(s, start, take_len);\n".to_string()), " s.chars().skip(start).take(take_len).collect()\n".to_string()), "}\n\n".to_string()), "pub fn substring(s: &str, start: i64, end: i64) -> String {\n".to_string()), " substring_ascii_aware(s, s.is_ascii(), start, end)\n".to_string()), "}\n\n".to_string()), "pub fn string_contains(s: &str, sub: String) -> bool {\n".to_string()), " s.contains(&*sub)\n".to_string()), "}\n\n".to_string()), "pub fn contains(s: String, sub: String) -> bool { string_contains(&s, sub) }\n\n".to_string()), "pub fn starts_with(s: String, prefix: String) -> bool { s.starts_with(&*prefix) }\n\n".to_string()), "pub fn ends_with(s: String, suffix: String) -> bool { s.ends_with(&*suffix) }\n\n".to_string()), "pub fn trim(s: String) -> String { s.trim().to_string() }\n\n".to_string()), "pub fn count(items: Rc>) -> i64 { items.len() as i64 }\n\n".to_string()), "pub fn str_eq(a: String, b: String) -> bool { a == b }\n\n".to_string()), "pub fn to_string(value: i64) -> String { value.to_string() }\n\n".to_string()), "/// RFC 3629 UTF-8 decode of a byte vector. Fail-closed on invalid UTF-8.\n".to_string()), "pub fn utf8_decode_bytes(bytes: &[u8]) -> Result {\n".to_string()), " String::from_utf8(bytes.to_vec()).map_err(|e| format!(\"invalid UTF-8 in access payload: {}\", e))\n".to_string()), "}\n\n".to_string()), "pub fn clamp(val: i64, min_val: i64, max_val: i64) -> i64 {\n".to_string()), " val.clamp(min_val, max_val)\n".to_string()), "}\n\n".to_string()) } pub fn rt_collection_ops() -> String { diff --git a/src/v1/stage0/src/v1_interpreter.rs b/src/v1/stage0/src/v1_interpreter.rs index 71de166cd79..b0a9cd7c60a 100644 --- a/src/v1/stage0/src/v1_interpreter.rs +++ b/src/v1/stage0/src/v1_interpreter.rs @@ -426,13 +426,93 @@ pub fn sorted_fields(mut v: Vec<(Symbol, Value)>) -> Vec<(Symbol, Value)> { v } +/// A shared string carrier that precomputes ASCII-ness once at construction (`RcStr::new`) +/// instead of rescanning on every `char_at`/`substring`/`string_length` call — the fact is +/// carried on the value (DESIGN §5 construction-over-validation), not looked up by an +/// identity inferred after the fact (the pointer-keyed cache class this replaces, withdrawn +/// in CHARAT-0 for aliasing on allocation reuse: STRING-INDEX-0). +#[derive(Debug, Clone)] +pub struct RcStr { + rc: Rc, + is_ascii: bool, +} + +impl RcStr { + pub fn new(rc: Rc) -> Self { + let is_ascii = rc.is_ascii(); + RcStr { rc, is_ascii } + } + + #[inline] + pub fn as_str(&self) -> &str { + &self.rc + } + + #[inline] + pub fn is_ascii(&self) -> bool { + self.is_ascii + } + + /// Owned clone of the underlying allocation, for sites that need a bare `Rc`. + pub fn rc(&self) -> Rc { + Rc::clone(&self.rc) + } + + pub fn ptr_eq(a: &RcStr, b: &RcStr) -> bool { + Rc::ptr_eq(&a.rc, &b.rc) + } +} + +impl std::ops::Deref for RcStr { + type Target = str; + fn deref(&self) -> &str { + &self.rc + } +} + +impl AsRef for RcStr { + fn as_ref(&self) -> &str { + &self.rc + } +} + +impl std::fmt::Display for RcStr { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + std::fmt::Display::fmt(&*self.rc, f) + } +} + +impl PartialEq for RcStr { + fn eq(&self, other: &Self) -> bool { + self.rc == other.rc + } +} +impl Eq for RcStr {} + +impl PartialOrd for RcStr { + fn partial_cmp(&self, other: &Self) -> Option { + Some(self.cmp(other)) + } +} +impl Ord for RcStr { + fn cmp(&self, other: &Self) -> std::cmp::Ordering { + self.rc.as_ref().cmp(other.rc.as_ref()) + } +} + +impl From> for RcStr { + fn from(rc: Rc) -> Self { + RcStr::new(rc) + } +} + #[derive(Debug, Clone)] pub enum Value { Null, Bool(bool), Int(i64), Float(f64), - Str(Rc), + Str(RcStr), List(Rc>), Map(Rc>), Set(Rc>), @@ -461,7 +541,7 @@ pub(crate) fn list_value(items: impl Into>) -> Value { } pub fn str_value(s: impl AsRef) -> Value { - Value::Str(Rc::from(s.as_ref())) + Value::Str(RcStr::new(Rc::from(s.as_ref()))) } /// Project an observed child-process status onto `std.process_termination` `ProcessTermination`. @@ -5638,7 +5718,7 @@ fn eval_cast(node: &Rc, env: &Rc, ctx: &InterpContext) -> InterpResul Value::Int(n) => Ok(str_value(n.to_string())), Value::Float(n) => Ok(str_value(n.to_string())), Value::Bool(b) => Ok(str_value(b.to_string())), - Value::Str(s) => Ok(Value::Str(Rc::clone(&s))), + Value::Str(s) => Ok(Value::Str(s.clone())), // Corpus wire/debug casts for structured values — not the blanket Display // fallback that silently stringified List/Map (§5 fabricated plausible output). Value::Variant { .. } | Value::Record { .. } => Ok(str_value(format!("{}", val))), @@ -5682,11 +5762,15 @@ fn eval_index(node: &Rc, env: &Rc, ctx: &InterpContext) -> InterpResu raw_map_lookup(base, key, env, ctx).map(RawMapLookup::into_raw) } (Value::Str(s), Value::Int(i)) => { - let i = *i as usize; - Ok(s.chars() - .nth(i) - .map(|c| str_value(c.to_string())) - .unwrap_or(Value::Null)) + if *i < 0 { + return Ok(Value::Null); + } + let ch = v1_rt::char_at_ascii_aware(s.as_str(), s.is_ascii(), *i); + if ch.is_empty() { + Ok(Value::Null) + } else { + Ok(str_value(ch)) + } } _ => Err(InterpError::TypeError { msg: format!( @@ -5714,10 +5798,21 @@ fn eval_slice(node: &Rc, env: &Rc, ctx: &InterpContext) -> InterpResu Ok(list_value(work.slice(s..e))) } (Value::Str(str_val), Value::Int(s), Value::Int(e)) => { - let s = *s as usize; - let e = *e as usize; - let sliced: String = str_val.chars().skip(s).take(e.saturating_sub(s)).collect(); - Ok(str_value(sliced)) + if *s >= 0 && *e >= 0 { + Ok(str_value(v1_rt::substring_ascii_aware( + str_val.as_str(), + str_val.is_ascii(), + *s, + *e, + ))) + } else { + // Negative indices wrap in the pre-carrier cast semantics (`*s as usize`); + // preserved verbatim off the ascii-aware fast path, which clamps to 0. + let s = *s as usize; + let e = *e as usize; + let sliced: String = str_val.chars().skip(s).take(e.saturating_sub(s)).collect(); + Ok(str_value(sliced)) + } } _ => Err(InterpError::TypeError { msg: format!( @@ -6210,17 +6305,31 @@ macro_rules! v1_algebra_method_arms { }, arm "method_call.substring" { "substring" } => { - let s = expect_string(&$receiver, "substring")?; + let s = expect_value_str(Some(&$receiver), "substring")?; match $args { [start, end] => { - let s_idx = expect_int(Some(start), "substring start")? as usize; - let e_idx = expect_int(Some(end), "substring end")? as usize; - let sliced: String = s - .chars() - .skip(s_idx) - .take(e_idx.saturating_sub(s_idx)) - .collect(); - Ok(str_value(sliced)) + let s_idx = expect_int(Some(start), "substring start")?; + let e_idx = expect_int(Some(end), "substring end")?; + if s_idx >= 0 && e_idx >= 0 { + Ok(str_value(v1_rt::substring_ascii_aware( + s.as_str(), + s.is_ascii(), + s_idx, + e_idx, + ))) + } else { + // Negative indices wrap in the pre-carrier cast semantics + // (`idx as usize`); preserved off the ascii-aware fast path, + // which clamps to 0 instead. + let s_idx = s_idx as usize; + let e_idx = e_idx as usize; + let sliced: String = s + .chars() + .skip(s_idx) + .take(e_idx.saturating_sub(s_idx)) + .collect(); + Ok(str_value(sliced)) + } } _ => Err(InterpError::TypeError { msg: "substring requires (start, end) arguments".to_string(), @@ -6229,12 +6338,17 @@ macro_rules! v1_algebra_method_arms { }, arm "method_call.char_at" { "char_at" } => { - let s = expect_string(&$receiver, "char_at")?; + let s = expect_value_str(Some(&$receiver), "char_at")?; let idx = expect_int($args.first(), "char_at")?; - Ok(s.chars() - .nth(idx as usize) - .map(|c| str_value(c.to_string())) - .unwrap_or(Value::Null)) + if idx < 0 { + return Ok(Value::Null); + } + let ch = v1_rt::char_at_ascii_aware(s.as_str(), s.is_ascii(), idx); + if ch.is_empty() { + Ok(Value::Null) + } else { + Ok(str_value(ch)) + } }, arm "method_call.index_by" { "index_by" } => list_method_with_closure( @@ -6941,7 +7055,7 @@ fn operation_input_binding_entry( } => { if *variant_name == ctx.sym("InputText") { match fields_get(fields, ctx.sym("text")).cloned() { - Some(Value::Str(text)) => Value::Str(Rc::clone(&text)), + Some(Value::Str(text)) => Value::Str(text.clone()), _ => { return Err(ArgvRefusalCause::BindingMalformed(format!( "InputText for `{name}` carries no String text" @@ -11397,21 +11511,32 @@ macro_rules! v1_builtin_arms { }, arm "free_call.string_length" { "string_length" } => { - let s = expect_str($positional.first().copied(), "string_length")?; - Ok(Some(Value::Int(s.chars().count() as i64))) + let s = expect_value_str($positional.first().copied(), "string_length")?; + Ok(Some(Value::Int(v1_rt::string_length_ascii_aware(s.as_str(), s.is_ascii())))) }, arm "free_call.substring" { "substring" } => { - let s = expect_str($positional.first().copied(), "substring")?; + // `v1_rt::substring` already clamps negative start/end to 0 (unlike the raw + // `as usize` casts at the other call sites), so the ascii-aware variant's + // identical `.max(0)` clamping preserves this arm's prior behavior exactly. + let s = expect_value_str($positional.first().copied(), "substring")?; let start = expect_int($positional.get(1).copied(), "substring start")?; let end = expect_int($positional.get(2).copied(), "substring end")?; - Ok(Some(str_value(v1_rt::substring(&s, start, end)))) + Ok(Some(str_value(v1_rt::substring_ascii_aware( + s.as_str(), + s.is_ascii(), + start, + end, + )))) }, arm "free_call.char_at" { "char_at" } => { - let s = expect_str($positional.first().copied(), "char_at")?; + // `v1_rt::char_at` already clamps a negative pos to 0 (unlike the raw + // `as usize` casts at the other call sites), so the ascii-aware variant's + // identical `.max(0)` clamping preserves this arm's prior behavior exactly. + let s = expect_value_str($positional.first().copied(), "char_at")?; let pos = expect_int($positional.get(1).copied(), "char_at pos")?; - Ok(Some(str_value(v1_rt::char_at(&s, pos)))) + Ok(Some(str_value(v1_rt::char_at_ascii_aware(s.as_str(), s.is_ascii(), pos)))) }, arm "free_call.string_contains" { "string_contains" } => { @@ -11432,7 +11557,10 @@ macro_rules! v1_builtin_arms { }, arm "free_call.length" { "length" } => match $positional.first() { - Some(Value::Str(s)) => Ok(Some(Value::Int(s.chars().count() as i64))), + Some(Value::Str(s)) => Ok(Some(Value::Int(v1_rt::string_length_ascii_aware( + s.as_str(), + s.is_ascii(), + )))), Some(v) => match native_len(v) { Some(n) => Ok(Some(Value::Int(n))), None => match free_monoid_to_vec(v) { @@ -13463,6 +13591,25 @@ fn expect_str(val: Option<&Value>, context: &str) -> InterpResult { } } +/// Like `expect_string`/`expect_str` but returns a reference into the `Value::Str` +/// payload instead of an owned copy — avoids the O(n) `.to_string()` allocation for +/// call sites that only need a `&str` view plus the carried ascii flag (STRING-INDEX-0). +fn expect_value_str<'a>(val: Option<&'a Value>, context: &str) -> InterpResult<&'a RcStr> { + match val { + Some(Value::Str(s)) => Ok(s), + Some(v) => Err(InterpError::TypeError { + msg: format!( + "{} expects a string argument, got {}", + context, + v.type_label() + ), + }), + None => Err(InterpError::TypeError { + msg: format!("{} requires a string argument", context), + }), + } +} + fn expect_byte_vec(val: Option<&Value>, context: &str) -> InterpResult> { match val { Some(Value::List(items)) => { @@ -14932,7 +15079,7 @@ mod value_str_rc_semantic_parity_tests { assert_ne!(a, c); if let (Value::Str(ra), Value::Str(rb)) = (&a, &b) { assert!( - !Rc::ptr_eq(ra, rb), + !RcStr::ptr_eq(ra, rb), "distinct Rc allocations must still compare equal by content" ); } @@ -14986,7 +15133,7 @@ mod value_str_rc_semantic_parity_tests { let cloned = v.clone(); if let (Value::Str(a), Value::Str(b)) = (&v, &cloned) { assert!( - Rc::ptr_eq(a, b), + RcStr::ptr_eq(a, b), "Value::Str clone must share the same Rc allocation (not deep-copy); \ content-equality tests alone would stay green if clone reintroduced String copy" ); @@ -15005,9 +15152,9 @@ mod value_str_rc_semantic_parity_tests { let Value::Str(b) = &cloned else { panic!("expected Str"); }; - assert!(Rc::ptr_eq(a, b)); - let mut lone = Rc::clone(a); - let mut peer = Rc::clone(b); + assert!(RcStr::ptr_eq(a, b)); + let mut lone = a.rc(); + let mut peer = b.rc(); assert!( Rc::get_mut(&mut lone).is_none(), "shared Rc must refuse in-place mutation while another handle exists" diff --git a/src/v1/stage0/src/v1_rt.rs b/src/v1/stage0/src/v1_rt.rs index f46ad0294f4..d83749ce800 100644 --- a/src/v1/stage0/src/v1_rt.rs +++ b/src/v1/stage0/src/v1_rt.rs @@ -274,9 +274,12 @@ pub fn concat(a: T, b: T) -> T { a.v1_concat(b) } -pub fn char_at(s: &str, pos: i64) -> String { +/// Ascii-aware variant taking a precomputed `is_ascii` flag (the `RcStr` carrier fact) +/// instead of rescanning the whole string on every call -- the per-call `s.is_ascii()` +/// scan is what made repeated indexing over a large string O(n^2) (STRING-INDEX-0). +pub fn char_at_ascii_aware(s: &str, is_ascii: bool, pos: i64) -> String { let pos = pos.max(0) as usize; - if s.is_ascii() { + if is_ascii { let bytes = s.as_bytes(); if pos >= bytes.len() { return String::new(); @@ -289,21 +292,31 @@ pub fn char_at(s: &str, pos: i64) -> String { .unwrap_or_default() } -pub fn string_length(s: &str) -> i64 { - if s.is_ascii() { +pub fn char_at(s: &str, pos: i64) -> String { + char_at_ascii_aware(s, s.is_ascii(), pos) +} + +/// Ascii-aware variant taking a precomputed `is_ascii` flag; see `char_at_ascii_aware`. +pub fn string_length_ascii_aware(s: &str, is_ascii: bool) -> i64 { + if is_ascii { s.len() as i64 } else { s.chars().count() as i64 } } -pub fn substring(s: &str, start: i64, end: i64) -> String { +pub fn string_length(s: &str) -> i64 { + string_length_ascii_aware(s, s.is_ascii()) +} + +/// Ascii-aware variant taking a precomputed `is_ascii` flag; see `char_at_ascii_aware`. +pub fn substring_ascii_aware(s: &str, is_ascii: bool, start: i64, end: i64) -> String { let start = start.max(0) as usize; let end = end.max(0) as usize; if end <= start { return String::new(); } - if s.is_ascii() { + if is_ascii { let len = s.len(); if start >= len { return String::new(); @@ -318,6 +331,10 @@ pub fn substring(s: &str, start: i64, end: i64) -> String { s.chars().skip(start).take(take_len).collect() } +pub fn substring(s: &str, start: i64, end: i64) -> String { + substring_ascii_aware(s, s.is_ascii(), start, end) +} + pub fn string_contains(s: &str, sub: String) -> bool { s.contains(&*sub) }