Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
58 commits
Select commit Hold shift + click to select a range
dcd62b9
std.unicode.scalar: declare partial from_code_point (typed NotUnicode…
Oct 5, 2026
4a6934b
RFM: bare from_code_point binds the total seed builtin (population at…
Oct 5, 2026
50c3724
wip
Oct 5, 2026
ab4e5b7
Merge remote-tracking branch 'origin/session/bright-fox-380-fcp' into…
Oct 5, 2026
57d73cc
std seam sites: trim and utf8_decode_bytes stop binding bare from_cod…
Oct 5, 2026
9b5cbe7
Parser decodes route NotUnicodeScalar into each parser's own typed re…
Oct 5, 2026
749fed4
XL-2 CHAR follow-up: migrate 79 Char-by-construction from_code_point …
Oct 5, 2026
6b910ff
Octet-as-char meaning fork: declared std.encoding ASCII route; rfc_52…
Oct 5, 2026
4320c18
witness: import the live-tree disposition explicitly (no bare channel)
Oct 5, 2026
7975d31
Record trim's in-place seam copy honestly and against the seam's diss…
Oct 5, 2026
5500919
fabric witness: import LiveTreeDisposition
Oct 5, 2026
fafb4bc
yaml ingest: match yaml_first_refused once, dropping the unreachable …
Oct 5, 2026
eb4267d
Merge remote-tracking branch 'origin/session/bright-fox-380-fcp' into…
Oct 5, 2026
09b7f3e
json_hex_nibble importers name the declarer (extdeps.languages.json.g…
Oct 5, 2026
9468f83
Delete the from_code_point seed builtin (v1 purpose test: v2 single a…
Oct 5, 2026
f7bda8a
wip-tmp
Oct 5, 2026
a771a34
Merge branch 'session/bold-deer-208' into session/bold-deer-208-utf8
Oct 5, 2026
f27f5d5
utf8_decode_octets yields List<Char>; four callers and uri spell text…
Oct 5, 2026
1cd5496
RFM: name the decoder-relocation follow-up as the uri duplicate's tri…
Oct 5, 2026
f4970a2
Explicit imports where the char_text import turned off the bare chann…
Oct 5, 2026
1075f5e
v1 tokenize: source_char crosses the index through source_code_point …
Oct 5, 2026
2b70fa6
witness: import the live-tree disposition explicitly (no bare channel)
Oct 5, 2026
07a21d9
git_ls_remote witness: import SubstrateInputsOnly (claims ran nothing…
Oct 5, 2026
d88ee05
RFM receipts: the builtin's silent U+0000 (with REDs); name-migrated-…
Oct 5, 2026
e322b35
Merge branch 'session/bright-fox-380-fcp' of https://github.com/gunb-…
Oct 5, 2026
92cdc50
Merge branch 'session/bold-deer-208' into session/bold-deer-208-utf8
Oct 5, 2026
2768bab
Retire the git_ls_remote SubstrateInputsOnly debt row: ImportsFixed
Oct 5, 2026
75490d1
Merge remote-tracking branch 'origin/session/sleek-fox-462' into sess…
Oct 5, 2026
146c612
Merge remote-tracking branch 'origin/session/nimble-wolf-584' into se…
Oct 5, 2026
1bc43e9
Merge origin/session/proud-crane-779 (std seams) into the from_code_p…
Oct 5, 2026
334b945
v1 runtime: delete v1_rt::from_code_point (its last bridge row is gon…
Oct 5, 2026
48adbb9
Merge #13378's current head (with the CHAR, octet and seam lanes) int…
Oct 5, 2026
ed8f41c
RFM: record that the interpreted (NUL) and emitted (empty) realizatio…
Oct 5, 2026
dce1fab
Merge the parser-decode lane (with #13378's current head) into the de…
Oct 5, 2026
7364012
RFM: state the row's description over time (found / intermediate / la…
Oct 5, 2026
b353578
Merge remote-tracking branch 'origin/session/bold-deer-208' into sess…
Oct 5, 2026
8511cab
git_upstream_model_witness: the ls-tree -z fixture's U+00FF is a Char…
Oct 5, 2026
210b582
Merge remote-tracking branch 'origin/session/bold-deer-208-utf8' into…
Oct 5, 2026
bb5c44b
Hoist two body annotations above their declarations (DESIGN 4c: only …
Oct 5, 2026
d29519f
Merge remote-tracking branch 'origin/session/nimble-wolf-584-ff-char'…
Oct 5, 2026
4005683
std.coercion: import the three types it names (NonEmptyStr, Declarati…
Oct 5, 2026
13ade3a
Merge remote-tracking branch 'origin/main' into session/bright-fox-38…
Oct 5, 2026
5c06cef
stage0 partition: place std.unicode.scalar in the std-core unit (besi…
Oct 5, 2026
86f9b97
Merge remote-tracking branch 'origin/main' into session/bright-fox-38…
Oct 5, 2026
0508907
Merge origin/main into the parser-decode lane
Oct 5, 2026
dae3da9
Migrate the two bare from_code_point callers main added after the cen…
Oct 5, 2026
a18fc32
Merge branch 'session/bright-fox-380-fcp' of https://github.com/gunb-…
Oct 5, 2026
14956bb
stage0: install the emitted std_unicode_scalar mirror and its lib.rs …
Oct 5, 2026
e38bca2
stage0: install the mirrors required-regen named (round 1)
Oct 5, 2026
4d93cea
stage0 std-core crate: declare std_unicode_scalar (byte-identical to …
Oct 5, 2026
7ac592f
stage0: install the mirrors required-regen named (lib.rs, v1_rt.rs: v…
Oct 5, 2026
ff24e98
stage0 partition: re-export std_unicode_scalar through the emit-core …
Oct 5, 2026
8604690
stage0: regen round 1 (partition re-export of std_unicode_scalar)
Oct 5, 2026
860c27a
stage0: regen round 2 (partition re-export of std_unicode_scalar)
Oct 5, 2026
9524c85
Merge remote-tracking branch 'origin/main' into session/bright-fox-38…
Oct 5, 2026
41f0948
stage0: regen round 1 on the merged head
Oct 5, 2026
5fd449c
stage0: regen round 2 on the merged head
Oct 5, 2026
eea9a7e
Merge commit '5fd449cc96e467fa911b36bde18a5cfc0ce035ec' into session/…
Oct 5, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion dag/extdeps/bmc/ipmi_master_write_read.dag
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ import std.decl_ref { DeclarationRef, WholeDeclaration }
import v2.std.optional { Present }
import extdeps.external_authority { ExternalAuthority, ExternalModelScope, ExternalSubjectRef }
import extdeps.uri { Uri, Https }
import extdeps.languages.json.parse { json_hex_nibble }
import extdeps.languages.json.grammar { json_hex_nibble }

data extdeps_external_authority_anchor: ExternalAuthority = ExternalAuthority {
uri: Uri {
Expand Down
3 changes: 2 additions & 1 deletion dag/extdeps/dns/domain_name.dag
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
module extdeps.dns.domain_name

import std.unicode.scalar { char_text }
import std.types { NonEmptyStr, String, List, Bool, Int, brand }
import std.error_primitives { Result, Ok, Err }

Expand Down Expand Up @@ -71,7 +72,7 @@ fn fold_dns_case_string_at(s: String, i: Int, acc: String) -> String {
let ch = char_at(s, i)
let cp = code_point_int(ch: ch)
let out = if cp >= 65 && cp <= 90 {
from_code_point(cp + 32)
char_text(c: cp + 32)
} else {
ch
}
Expand Down
3 changes: 2 additions & 1 deletion dag/extdeps/git/git.dag
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
module extdeps.git

import std.unicode.scalar { char_text }
import extdeps.exec.program { ProgramIdentity, uncataloged_program }

import extdeps.external_authority { ExternalAuthority }
Expand Down Expand Up @@ -136,7 +137,7 @@ type GitDiffNameStatusParseState {
entries: List<GitDiffStatusEntry>
}

data git_diff_name_status_field_separator: String = from_code_point(0)
data git_diff_name_status_field_separator: String = char_text(c: 0)

fn git_diff_change_status_from_token(token: String) -> GitDiffChangeStatus {
let letter = substring(s: token, start: 0, end: 1)
Expand Down
20 changes: 11 additions & 9 deletions dag/extdeps/git/object_store.dag
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
module extdeps.git.object_store

import std.unicode.scalar { char_text }
import std.algebra { Cons, trim }

import std.types { NonEmptyStr, Bytes, Map, brand, range, Bool, Int, List }
Expand All @@ -20,6 +21,7 @@ import std.integer { UInt8, NonNegativeInt, PositiveInt, qualified_octet_members
import std.render_repeat_string_bootstrap { repeat_string }
import std.measure { ByteSize, byte_size, byte_size_count }
import std.bytes { bytes_octets, octets_bytes, utf8_encode_bytes }
import std.encoding { ascii_decode_octets, AsciiDecoded, AsciiNonAscii }
import std.cache_interface { ContentAddressedByValue, WriteOnce }
import extdeps.git { GitAuthor, ObjectType, BlobObj, TreeObj, CommitObj, TagObj }
import extdeps.object_storage { ObjectContainer }
Expand Down Expand Up @@ -928,7 +930,7 @@ fn git_store_object_canonical_hash_input(
),
" ",
to_string(count(payload_octets)),
from_code_point(0),
char_text(c: 0),
],
"",
)
Expand Down Expand Up @@ -1974,16 +1976,16 @@ type GitLsTreeZDecodeOutcome
= GitLsTreeZDecoded { entries: List<GitTreeWireEntry> }
| GitLsTreeZRefused { cause: GitLsTreeZRefusal }

// A git NUL-framed header field is ASCII text through the declared std.encoding ascii_decode_octets
// route; NUL is ASCII but is the field DELIMITER here, so it is refused by this field's own policy.
fn git_ascii_field_text(octets: List<UInt8>) -> NonEmptyStr? {
if all(octets, octet => octet > 0 && octet < 128) {
Present {
value: join(
octets |> map(octet => from_code_point(octet)),
"",
) as NonEmptyStr,
}
} else {
if !all(octets, octet => octet != 0) {
none
} else {
match ascii_decode_octets(octets: octets) {
AsciiDecoded { text } => Present { value: text as NonEmptyStr }
AsciiNonAscii { at: _, octet: _ } => none
}
}
}

Expand Down
3 changes: 2 additions & 1 deletion dag/extdeps/github/actions.dag
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
module extdeps.github.actions

import std.unicode.scalar { char_text }
import std.decl_ref { DeclarationRef, decl_ref }
import extdeps.cron.schedule_model { CronSchedule }
import extdeps.github.log_annotations {
Expand Down Expand Up @@ -342,7 +343,7 @@ fn runner_arch_label(arch: Architecture) -> String {
// DIFFERENT labels, which surfaces as a loud divergence rather than a silent match.
fn runner_label_match_key(label: String) -> String {
join(
label |> chars |> map(c => if c >= 65 && c <= 90 { from_code_point(cp: c + 32) } else { from_code_point(cp: c) }),
label |> chars |> map(c => if c >= 65 && c <= 90 { char_text(c: c + 32) } else { char_text(c: c) }),
"",
)
}
Expand Down
3 changes: 2 additions & 1 deletion dag/extdeps/http/form_urlencoded.dag
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@ import std.types { String, List, Bool, Int }
import std.integer { UInt8 }
import std.bytes { utf8_encode_bytes, bytes_octets }
import std.encoding { utf8_decode_octets }
import std.coercion { unicode_scalar_fold }
import v2.std.algebra { any, filter }

// WHATWG URL, application/x-www-form-urlencoded: split pairs before decoding, plus means space,
Expand Down Expand Up @@ -47,7 +48,7 @@ fn form_component(wire: String) -> String? {
FormOctetsRefused => none
FormOctetsDecoded { octets } => match utf8_decode_octets(octets: octets) {
Absent => none
Present { value: scalars } => Present { value: join(map(scalars, cp => from_code_point(cp: cp)), "") }
Present { value: scalars } => Present { value: unicode_scalar_fold(xs: scalars) }
}
}
}
Expand Down
3 changes: 2 additions & 1 deletion dag/extdeps/languages/json/emit.dag
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
module extdeps.languages.json.emit

import std.unicode.scalar { char_text }
import extdeps.external_authority { ExternalAuthority }
import extdeps.languages.json.grammar { JsonNumberLexeme, json_int_lexeme, json_number_lexeme }
import extdeps.numeric.base16 { int_to_upper_hex }
Expand Down Expand Up @@ -101,7 +102,7 @@ data json_control_escape_code_points: List<Int> = [
]

fn json_escape_replace(s: String, code_point: Int, replacement: String) -> String {
join(split(s: s, delimiter: from_code_point(code_point)), replacement)
join(split(s: s, delimiter: char_text(c: code_point)), replacement)
}

fn json_escape_named_replacement(code_point: Int) -> String {
Expand Down
120 changes: 7 additions & 113 deletions dag/extdeps/languages/json/parse.dag
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,6 @@ import extdeps.languages.json.grammar {
JsonTextParseResult, JsonTextParseOk, JsonTextParseFail,
json_text_skip_ws, json_text_parse_string, json_text_parse_number,
json_number_lexeme_unchecked, is_json_digit_char,
json_hex_nibble, json_hex4_at,
}

// parse_json is the FORWARD reading of RFC 8259: text -> JsonValue, the inverse of
Expand All @@ -25,17 +24,12 @@ import extdeps.languages.json.grammar {
// the escape SET (one native split per string body, RFC 8259 section 7), so an unknown escape or
// a short \u refuses the parse before a value is built and never reaches the unescaper below -
// see json_escape_production_note for why that refusal lives at the scan and what it prevents
// (review 45642). The malformed-\u arm in json_unescape_decode_piece is therefore unreachable for
// any span this parser accepts; it is retained as a total fallback because json_unescape is
// exported and a caller could hand it an unvalidated span. \uXXXX IS decoded to its code point,
// which is what makes the round-trip total against escape_json_string: that emitter maps code
// points 0-31 other than the six with short escapes to \uXXXX, so a parser that passed \u through
// would silently corrupt every control character a receipt detail field can carry. Surrogate
// PAIRS are not recombined - serialize_json never emits one (it escapes only 0-31), so a lone
// \uXXXX decode is exact over this emitter's output; foreign JSON carrying an astral character as
// a surrogate pair is outside the declared fragment. A malformed \u (fewer than four hex digits,
// or a non-hex digit) refuses at the scan, so json_hex4_at's negative arm is likewise unreachable
// from parse_json and kept only for the exported-caller case.
// (review 45642). \uXXXX IS decoded to its scalar by the native json_unescape_checked kernel,
// which recombines a UTF-16 surrogate pair (RFC 8259 section 7) and refuses an unpaired surrogate,
// so a value this parser builds never carries a surrogate code point. There is no interpreted
// unescaper beside it: the former json_unescape piece decoder had no consumer and spelled each
// \u through the total seed from_code_point (gunbc.recurring_failure_mode
// bare_from_code_point_binds_the_total_seed_builtin), so it was deleted rather than migrated.

data extdeps_external_authority_anchor: ExternalAuthority = ExternalAuthority {
uri: Uri {
Expand All @@ -48,111 +42,11 @@ type JsonParse =
JsonParsed { value: JsonValue, next: Int }
| JsonParseFail { at: Int }

// THE UNESCAPER MOVES IN ESCAPE-SIZED STEPS, NOT PER CHARACTER (measured 2026-09-23 on
// corpus-scale shards: even with the span unescape, the per-character next-escape scan put the
// parse at roughly fourteen interpreter-minutes per megabyte, which is the 104 MB envelope's
// wall). One native split on the backslash resolves the whole escape structure — the pieces'
// boundaries ARE the parity answer, so `\\` and `\u` sequences cannot be misparsed the way
// sequential replacement passes would misparse them — and each piece decodes in one interpreted
// step. The decoded pieces stay in document order (the recursion carries the first piece
// first) and the join is one native pass; neither side copies an accumulator per escape.
// THE FIRST PIECE IS LITERAL BY CONSTRUCTION (it precedes the first backslash); the rest are
// escapes. The split is walked once, head by head, and each decoded piece is carried in
// document order.
fn json_unescape_decoded_pieces(pieces: List<String>) -> List<String> {
match pieces.first() {
Absent => []
Present { value: head } =>
concat(
[head],
json_unescape_decode_rest(pieces: json_unescape_drop_first(xs: pieces), literal_next: false),
)
}
}

// THE WALK CARRIES THE PARITY ANSWER: an empty piece is a \\ pair, and the piece after a pair
// is LITERAL — its head was never an escape letter, because the backslash that would have made
// it one is the pair's second half. (Without the flag, `\\b` decodes the b as \b and `\q`
// refuses a valid document.) The flag lasts exactly one piece.
// ONE PASS, THE PARITY AND THE DECODED PIECES IN THE ACCUMULATOR: the recursive walk
// re-dropped the tail at every level (and the drop concatenated at the end, copying per
// element), the square of a string's escape count — the read's long pole at the project
// envelope's ~3.45M escapes. The fold visits each piece once, prepends, and reverses once;
// the state machine is the recursion's, so the decoded bytes are unchanged.
type JsonUnescapeDecodeAcc { literal_next: Bool, decoded_rev: List<String> }

fn json_unescape_decode_rest(pieces: List<String>, literal_next: Bool) -> List<String> {
let acc = fold(
pieces,
init: JsonUnescapeDecodeAcc { literal_next: literal_next, decoded_rev: [] },
f: fn(a, piece) {
if a.literal_next {
JsonUnescapeDecodeAcc { literal_next: false, decoded_rev: concat([piece], a.decoded_rev) }
} else if piece == "" {
JsonUnescapeDecodeAcc { literal_next: true, decoded_rev: concat(["\\"], a.decoded_rev) }
} else {
JsonUnescapeDecodeAcc {
literal_next: false,
decoded_rev: concat([json_unescape_decode_piece(piece: piece)], a.decoded_rev),
}
}
},
)
reverse(acc.decoded_rev)
}

type JsonUnescapeDropAcc { skipped: Bool, kept: List<String> }

fn json_unescape_drop_first(xs: List<String>) -> List<String> {
let result = fold(xs, init: JsonUnescapeDropAcc { skipped: false, kept: [] }, f: fn(acc, x) {
if acc.skipped { JsonUnescapeDropAcc { skipped: true, kept: concat([x], acc.kept) } } else { JsonUnescapeDropAcc { skipped: true, kept: [] } }
})
reverse(result.kept)
}

// AN EMPTY PIECE IS THE ESCAPED BACKSLASH; any other piece's head is its escape letter, with \u
// carrying exactly four hex digits. The malformed-\u arm is the exported-caller fallback (the
// grammar refuses before value building, so a short or non-hex \u never reaches this builder
// from the parser): it drops the backslash and rescans the remainder as literal text, the
// total-fallback semantics, piece-local.
fn json_unescape_decode_piece(piece: String) -> String {
if piece == "" {
"\\"
} else {
let c = char_at(s: piece, pos: 0)
if c == "u" {
let cp = json_hex4_at(s: piece, i: 1)
if cp < 0 {
concat("u", json_unescape(s: substring(s: piece, start: 1, end: piece.length())))
} else {
concat(from_code_point(cp: cp), substring(s: piece, start: 5, end: piece.length()))
}
} else {
let u = if c == "n" { from_code_point(cp: 10) }
else if c == "t" { from_code_point(cp: 9) }
else if c == "r" { from_code_point(cp: 13) }
else if c == "b" { from_code_point(cp: 8) }
else if c == "f" { from_code_point(cp: 12) }
else { c }
concat(u, substring(s: piece, start: 1, end: piece.length()))
}
}
}

fn json_unescape(s: String) -> String {
if string_contains(s: s, pattern: "\\") {
join(json_unescape_decoded_pieces(pieces: split(s: s, delimiter: "\\")), "")
} else {
s
}
}

// THE VALUE IS THE NATIVE json_unescape_checked DECODE: the span was validated by the grammar
// production one call ago, so the kernel's refusal arm is unreachable here — it is matched
// anyway, because a total path outlives the argument for why it is dead. This replaces the
// interpreted piece decode (one interpreted step per escape over the project envelope's ~3.45M
// escapes, the read's measured wall) with the same bytes at native speed; json_unescape below
// stays as the exported-caller authority, malformed-\u fallback included.
// escapes, the read's measured wall) with the same bytes at native speed.
fn parse_json_string_at(s: String, i: Int) -> JsonParse {
match json_text_parse_string(s: s, i: i) {
JsonTextParseFail { end: e } => JsonParseFail { at: e }
Expand Down
1 change: 0 additions & 1 deletion dag/extdeps/languages/rust/emit.dag
Original file line number Diff line number Diff line change
Expand Up @@ -287,7 +287,6 @@ data rt_function_registry: List<RuntimeFunction> = [
{ name: "scan_to_eol", bridge_name: "scan_to_eol", passes_by_ref: true, wraps_result: false },
{ name: "scan_string_end", bridge_name: "scan_string_end", passes_by_ref: true, wraps_result: false },
{ name: "code_point", bridge_name: "code_point", passes_by_ref: false, wraps_result: false },
{ name: "from_code_point", bridge_name: "from_code_point", passes_by_ref: false, wraps_result: false },
{ name: "lookup", bridge_name: "lookup", passes_by_ref: true, wraps_result: false },
{ name: "get", bridge_name: "list_get_optional", passes_by_ref: true, wraps_result: false },
{ name: "index_by", bridge_name: "rc_index_by", passes_by_ref: false, wraps_result: false },
Expand Down
5 changes: 3 additions & 2 deletions dag/extdeps/languages/toml/emit.dag
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
module extdeps.languages.toml.emit

import std.unicode.scalar { char_text }
import std.types { Bool, String, List, Int }
import v2.std.collection { list_at_optional }
import v2.std.algebra { list_head, skip, HeadFound, HeadAbsent }
Expand Down Expand Up @@ -77,7 +78,7 @@ fn toml_escape_remaining_control(point: Int) -> String {
if (point < 32) || (point == 127) {
concat("\\u00", concat(toml_hex_nibble(n: point / 16), toml_hex_nibble(n: point - ((point / 16) * 16))))
} else {
from_code_point(point)
char_text(c: point)
}
}

Expand Down Expand Up @@ -152,7 +153,7 @@ fn toml_comment_lines(text: String) -> List<String> {
// rather than merely verbose.
fn toml_comment_char(point: Int) -> String {
if point == 9 {
from_code_point(point)
char_text(c: point)
} else {
toml_escape_remaining_control(point: point)
}
Expand Down
3 changes: 2 additions & 1 deletion dag/extdeps/languages/yaml/emit.dag
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
module extdeps.languages.yaml.emit

import std.unicode.scalar { char_text }
import extdeps.external_authority { ExternalAuthority }
import extdeps.uri { Uri, Https }
import std.algebra { trim }
Expand Down Expand Up @@ -324,7 +325,7 @@ fn yaml_emitted_text(text: String) -> YamlEmitResult {
if yaml_contains_refused_character(text: text) {
let cp = yaml_first_refused_character(text: text)
YamlEmitRefused {
path: join(["line ", to_string(yaml_line_containing(lines: yaml_source_lines(src: text), pattern: from_code_point(cp: cp), index: 0)), " of the emitted text"], ""),
path: join(["line ", to_string(yaml_line_containing(lines: yaml_source_lines(src: text), pattern: char_text(c: cp), index: 0)), " of the emitted text"], ""),
reason: yaml_refused_character_reason(cp: cp),
}
} else {
Expand Down
Loading
Loading