Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
module gunbc.recurring_failure_mode.octet_offset_consumed_by_a_scalar_indexed_text_operation

import std.types { NonEmptyStr }
import std.decl_ref { DeclarationRef, WholeDeclaration }
import gunbc.recurring_failure_mode { RecurringFailureMode }

data octet_offset_consumed_by_a_scalar_indexed_text_operation: RecurringFailureMode = RecurringFailureMode {
identity: "octet_offset_consumed_by_a_scalar_indexed_text_operation" as NonEmptyStr,

receipts: [
"INVALID STATE: an offset measured in UTF-8 OCTETS is handed to a text operation that indexes Unicode SCALARS. The v2 tokenizer counts every walk offset in octets (v2.compiler.tokenize lexeme_utf8_octet_count), so every v2.std.diagnostic ByteRange the span index resolves to is an octet range. `substring` and `string_length` index scalars: the emitted runtime walks chars(). Both are plain Int, so nothing stops an octet offset from reaching substring. On ASCII text the two agree, which is why the defect passes every ASCII fixture and is silent on the first multibyte character.",
"SPECIMEN, CAUGHT IN REVIEW BEFORE LANDING (gunbc#13051, calm-boar-904): v2.cli.if_arm_reader_differential derived each site's line as the count of line feeds in substring(source, 0, ByteRange.start). Every multibyte scalar before a site moved its reported line. The census COUNT was unaffected; only the rendered locations were wrong. Repaired by reporting the byte range alone. The corpus's one line-from-octet reader was private to a lens (v2.lens.text_string_importer_census newline_offsets), not a shared authority, and swift-lynx-592 is moving it into v2.std.source_position. When that lands it is the sanctioned conversion this row's trigger names. The control is a nested if after 2-, 3- and 4-octet scalars, located at octet 237, which is scalar 223.",
"POPULATION, read at the same revision over dag/ and src/v2/ (about 486 substring calls, each offset traced to its producer): no other site passes an octet offset to substring. Two neighbours look like the class but are not. src/v2/lens/text_string_importer_census.dag newline_offsets counts octets over utf8_encode_bytes and compares them with ByteRange.start: octets against octets. dag/std/import.dag import_strip_step slices with import-statement spans from the v1 tokenizer, which indexes its source_chars code-point list: scalars against scalars. One site agrees only by its input: dag/gunbc/runner/runner_microvm.dag jit_drive_content subtracts an octet count from a substring end over a string of line feeds only, where octets and scalars coincide.",
"RUNG FOUND AT: below the floor (a plausible, wrong location rendered with no refusal). RUNG NOW: no wall. The population is zero by a one-off read, not by any gate, so a new site can be written today. CEILING: structurally impossible. An octet offset and a scalar index are different quantities and can be different types. NEXT-RUNG TRIGGER: v2.std.diagnostic ByteRange carries a distinct octet-offset type that substring does not accept, with a sanctioned conversion that walks the UTF-8 table (std.unicode.types unicode_scalar_utf8_octet_count). A source-locus renderer that wants lines then derives them from octets through that one conversion.",
],

evidence: [
DeclarationRef { module_path: "v2.test.claim.cli.if_arm_reader_differential", decl_name: "multibyte_source_before_a_site_is_located_in_octets_holds", field: WholeDeclaration },
DeclarationRef { module_path: "v2.compiler.tokenize", decl_name: "lexeme_utf8_octet_count", field: WholeDeclaration },
DeclarationRef { module_path: "v2.cli.if_arm_reader_differential", decl_name: "iard_location", field: WholeDeclaration },
],
}
86 changes: 86 additions & 0 deletions src/v2/cli/compile_cli.dag
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,13 @@ import v2.compiler.self_host.closure_emission {
emit_closure_from_ingest_located
}
import v2.compiler.source_authority { SourceRootIngest, source_root_ingest_from_files }
import v2.cli.if_arm_reader_differential {
IardCensus,
IardCensusTaken,
IardGrammarUnprepared,
if_arm_reader_differential,
if_arm_reader_differential_text
}
import gunbc.source_root_read { source_root_read, SourceRootFilesRead, SourceRootFilesRefused }
import v2.extdeps.languages.c { c_target_model }
import v2.extdeps.languages.rust { rust_target_model }
Expand Down Expand Up @@ -135,6 +142,7 @@ fn cli_emit_target_roster_text() -> String {

type CliVerb
= CliEmitClosure { entry: String, source_roots: FreeMonoid<String>, target: CliEmitTarget }
| CliIfArmReaderDifferential { source_roots: FreeMonoid<String> }

// A PLAN IS PARSED OR IT IS REFUSED, and the refusal carries its own reason symbol rather than a
// rendered sentence, so the host renders it once and a witness can discriminate on the CAUSE. A
Expand All @@ -156,6 +164,8 @@ type CliParseState
| CliParseSeekRootValue { entry: String, roots: FreeMonoid<String>, target: CliEmitTarget }
| CliParseSeekEntryValue { entry: String, roots: FreeMonoid<String>, target: CliEmitTarget }
| CliParseSeekTargetValue { entry: String, roots: FreeMonoid<String> }
| CliParseSeekDifferentialOption { roots: FreeMonoid<String> }
| CliParseSeekDifferentialRootValue { roots: FreeMonoid<String> }
| CliParseFailed { reason: Symbol, detail: String }

data cli_option_source_root: String = "--source-root"
Expand All @@ -166,21 +176,48 @@ data cli_option_target: String = "--target"

data cli_verb_emit: String = "emit"

// THE NESTED-IF ARM-READER DIFFERENTIAL (v2.cli.if_arm_reader_differential). Named for exactly what
// it measures so it does not read as the general census, which lives on the native driver
// (v2.compiler.compile). It takes source roots only, and is deleted with its module.
data cli_verb_if_arm_reader_differential: String = "if-arm-reader-differential"

data cli_usage: String = "usage: <binary> emit --entry <module.path> --source-root <dir> [--source-root <dir>]... [--target <target>]"

data cli_differential_usage: String = "usage: <binary> if-arm-reader-differential --source-root <dir> [--source-root <dir>]..."

fn cli_parse_step(state: CliParseState, arg: String) -> CliParseState {
match state {
CliParseFailed { reason: reason, detail: detail } =>
CliParseFailed { reason: reason, detail: detail }
CliParseSeekVerb =>
if arg == cli_verb_emit {
CliParseSeekOption { entry: "", roots: Empty, target: cli_default_emit_target }
} else if arg == cli_verb_if_arm_reader_differential {
CliParseSeekDifferentialOption { roots: Empty }
} else {
CliParseFailed {
reason: ^cli_unknown_verb,
detail: concat(concat("unknown verb ", arg), concat(" -- ", cli_usage))
}
}
CliParseSeekDifferentialOption { roots: roots } =>
if arg == cli_option_source_root {
CliParseSeekDifferentialRootValue { roots: roots }
} else {
CliParseFailed {
reason: ^cli_unknown_option,
detail: concat(concat("unknown option ", arg), concat(" -- ", cli_differential_usage))
}
}
CliParseSeekDifferentialRootValue { roots: roots } =>
if cli_arg_is_option(arg: arg) {
CliParseFailed {
reason: ^cli_source_root_missing_value,
detail: concat(cli_option_source_root, " takes a directory, and an option followed it")
}
} else {
CliParseSeekDifferentialOption { roots: Cons { head: arg, tail: roots } }
}
CliParseSeekRootValue { entry: entry, roots: roots, target: target } =>
if cli_arg_is_option(arg: arg) {
CliParseFailed {
Expand Down Expand Up @@ -263,6 +300,20 @@ fn cli_parse_finish(state: CliParseState) -> CliPlan {
reason: ^cli_target_missing_value,
detail: concat(cli_option_target, " takes a target name, and none followed it")
}
CliParseSeekDifferentialRootValue { roots: _ } =>
CliPlanRefused {
reason: ^cli_source_root_missing_value,
detail: concat(cli_option_source_root, " takes a directory, and none followed it")
}
CliParseSeekDifferentialOption { roots: roots } =>
if length(xs: roots) == 0 {
CliPlanRefused {
reason: ^cli_no_source_root,
detail: concat("if-arm-reader-differential needs at least one source root -- ", cli_differential_usage)
}
} else {
CliPlanParsed { verb: CliIfArmReaderDifferential { source_roots: list_reverse(xs: roots) } }
}
CliParseSeekOption { entry: entry, roots: roots, target: target } =>
if length(xs: roots) == 0 {
CliPlanRefused {
Expand Down Expand Up @@ -300,6 +351,7 @@ fn v2_cli_plan_source_roots(plan: CliPlan) -> FreeMonoid<String> {
CliPlanParsed { verb: verb } =>
match verb {
CliEmitClosure { entry: _, source_roots: roots, target: _ } => roots
CliIfArmReaderDifferential { source_roots: roots } => roots
}
}
}
Expand All @@ -309,6 +361,7 @@ type CliRunOutcome
| CliRunRefused { reason: Symbol, detail: String }
| CliCensusRefused { reason: Symbol, detail: String, text: String }
| CliUsageRefused { reason: Symbol, detail: String }
| CliDifferentialReported { text: String, observed: Bool, detail: String }

// THE REFUSAL PRINTS ITS CAUSE, AND FOR ONE GENERATION IT DID NOT. The rejected arm already
// CARRIED `d.head.reason` (now the FATAL reason -- the head was the first advisory that happened
Expand Down Expand Up @@ -349,6 +402,15 @@ fn v2_cli_run(plan: CliPlan, ingest: SourceRootIngest) -> CliRunOutcome {
CliUsageRefused { reason: reason, detail: detail }
CliPlanParsed { verb: verb } =>
match verb {
CliIfArmReaderDifferential { source_roots: _ } =>
if length(xs: ingest) == 0 {
CliRunRefused {
reason: ^cli_empty_ingest,
detail: "the source roots walked to zero .dag files; a census over nothing is not a census of zero"
}
} else {
cli_if_arm_reader_differential_outcome(census: if_arm_reader_differential(ingest: ingest))
}
CliEmitClosure { entry: entry, source_roots: _, target: target } =>
if length(xs: ingest) == 0 {
CliRunRefused {
Expand Down Expand Up @@ -393,6 +455,27 @@ fn v2_cli_run(plan: CliPlan, ingest: SourceRootIngest) -> CliRunOutcome {
}
}

// THE DIFFERENTIAL'S STATUS IS WHETHER THE POPULATION WAS WHOLLY OBSERVED, NEVER HOW MANY SITES
// DIFFER: the differing sites are the finding, not a failure. A file the census could not parse, or a
// grammar that did not prepare, leaves the population incomplete, and that fails the process so a
// partial count cannot be read as the whole one.
fn cli_if_arm_reader_differential_outcome(census: IardCensus) -> CliRunOutcome {
match census {
IardGrammarUnprepared { reason: _ } =>
CliDifferentialReported {
text: if_arm_reader_differential_text(c: census),
observed: false,
detail: "the grammar did not prepare; no site was observed"
}
IardCensusTaken { tally: t } =>
CliDifferentialReported {
text: if_arm_reader_differential_text(c: census),
observed: t.files_refused == 0,
detail: "some files did not parse; their sites are unobserved (see file_refused rows)"
}
}
}

// ONE RENDERING OF A LOCATED REFUSAL, for a refused closure and for a refused member alike: the whole
// ordered chain, then the fatal's file and extent. A member's refusal fails the process as a closure
// refusal does, so its status line owes the same located cause and not a summary sentence.
Expand Down Expand Up @@ -437,6 +520,8 @@ fn v2_cli_exit(outcome: CliRunOutcome) -> ProcessExit {
CliCensusRefused { reason: _, detail: detail, text: _ } => exit_failure(reason: detail)
CliUsageRefused { reason: _, detail: detail } =>
ExitFailure { code: exit_code_misuse, reason: detail }
CliDifferentialReported { text: _, observed: observed, detail: detail } =>
if observed { ExitSuccess } else { exit_failure(reason: detail) }
}
}

Expand All @@ -446,6 +531,7 @@ fn v2_cli_outcome_text(outcome: CliRunOutcome) -> String {
CliRunRefused { reason: _, detail: _ } => ""
CliCensusRefused { reason: _, detail: _, text: text } => text
CliUsageRefused { reason: _, detail: _ } => ""
CliDifferentialReported { text: text, observed: _, detail: _ } => text
}
}

Expand Down
Loading