diff --git a/dag/gunbc/namespace/namespace_step0_canary.dag b/dag/gunbc/namespace/namespace_step0_canary.dag index 53ceb8d5703..cc4f68df02e 100644 --- a/dag/gunbc/namespace/namespace_step0_canary.dag +++ b/dag/gunbc/namespace/namespace_step0_canary.dag @@ -199,6 +199,56 @@ fn namespace_step0_unavailable_receipt( } } +// THE STEP-0 PRODUCER'S OWN CONSTRUCTOR, and the one the freeze note above reserved for it: "the +// eventual strip producer must own a separate constructor whose inputs are the executed rewrite and +// its observation". Its inputs are exactly that -- the statements a typed source rewrite actually +// removed, and the cause the containment compile of the rewritten sources actually reported. It +// cannot be reached from the fixtures' caller seal and the fixtures cannot be reached from here, so +// production and fixture mint through disjoint doors and neither can stand in for the other. +// +// IT IS DECLARED HERE BECAUSE NamespaceStep0CanaryReceipt IS sole_constructor. Ownership of the +// production route is expressed by the caller seal, not by which file the constructor sits in: the +// only caller admitted is the producer's fold, so no other module -- and no future witness -- can +// mint a subject-entered receipt. +// +// THE ARGUMENTS IT REFUSES TO INVENT. A producer that stripped nothing has no observation, and +// passing an empty list here would mint a receipt the classifier refuses at ObservedSubjectField +// rather than one that quietly claims entry; the producer refuses before reaching this row instead. +// Likewise a containment compile that COMPLETED has no blocking cause, and there is no argument +// here that could express it -- which is why the producer's refusal coproduct carries that case. +fn namespace_step0_strip_applied_receipt( + contract_identity: NonEmptyStr, + source_commit: GitObjectId, + source_tree: GitObjectId, + compiler_program: NonEmptyStr, + compiler_executable_sha256: Sha256Digest, + input_subject_manifest: NonEmptyStr, + observed_tree: GitObjectId, + observed_import_statements: List, + containment_compile_cause: NonEmptyStr +) -> NamespaceStep0CanaryReceipt + admit_callers: [ + decl_ref(module_path: "gunbc.namespace_step0_strip_producer", decl_name: "namespace_step0_strip_receipt") + ] + = NamespaceStep0CanaryReceipt { + contract_identity: contract_identity, + source_commit: source_commit, + source_tree: source_tree, + compiler_program: compiler_program, + compiler_executable_sha256: compiler_executable_sha256, + input_subject_manifest: input_subject_manifest, + completed_production_prefix: [CanaryContractFrozen, SourceTreePinned, CompilerExecutablePinned, InputSubjectManifestPinned, TypedImportStripApplied], + first_blocking_transition: CompileContainmentResolution, + first_blocking_cause: ContainmentCompileObservationRefused { cause: containment_compile_cause }, + subject_entry: Step0SubjectEntered, + observation_completeness: Step0ObservationIncomplete, + observed_subject: Step0StripObservation { + observed_tree: observed_tree, + observed_manifest: input_subject_manifest, + observed_import_statements: observed_import_statements + } + } + // Admission is intentionally exact and narrow, but it is NOT frozen to today's frontier. The two // coherent shapes are today's stop at the strip and the next possible stop after a real strip has // entered the subject. That makes a longer prefix and a different blocker fixture-authorable now, diff --git a/dag/gunbc/namespace/namespace_step0_strip_producer.dag b/dag/gunbc/namespace/namespace_step0_strip_producer.dag new file mode 100644 index 00000000000..346b4257066 --- /dev/null +++ b/dag/gunbc/namespace/namespace_step0_strip_producer.dag @@ -0,0 +1,225 @@ +module gunbc.namespace_step0_strip_producer + +import std.types { Int, List, NonEmptyStr, String } +import std.content_hash { Sha256Digest } +import std.import { + ImportStripOutcome, + ImportStatementsStripped, + ImportStripRefused, + ImportStripRefusal, + ParsedImportStatement, + ParsedImportStatements, + ImportStatementsParsed, + ImportStatementParseRefused, + strip_import_statements, +} +import extdeps.git.object_store { GitObjectId } +import gunbc.parsed_import_statement_production { parsed_import_statements_live } +import gunbc.compile_diagnostic_census { CompileDiagnosticCensusRow } +import tools.multi_module_compile_fixture { + FixtureSource, + MultiModuleCompileFixture, + MultiModuleCompileFixtureOutcome, + FixtureInstrumentRefused, + FixtureCompileRefused, + FixtureCompileCompleted, + compile_fixture, +} +import gunbc.namespace_step0_canary { + NamespaceStep0CanaryReceipt, + namespace_step0_canary_contract, + namespace_step0_manifest_identity, + namespace_step0_strip_applied_receipt, +} + +// THE OWNED STEP-0 PRODUCER. It applies the typed import strip to a supplied subject and folds the +// compile observation that follows it, and those two acts are the two capabilities the frozen canary +// names as missing (gunbc.namespace_step0_canary namespace_step0_unavailable_receipt: "a typed +// parse-level source rewrite that removes one complete import statement" and "an owned Step-0 +// producer that applies the rewrite and folds the resulting compile observation"). +// +// EVERY STAGE REFUSES RATHER THAN WIDENS. A source whose imports did not parse is not a source with +// no imports; a span set the rewrite would not accept is not a clean rewrite; a subject in which +// nothing was removed did not enter the Step 0 subject. Each is a typed, located cause here, and +// none of them can reach the receipt constructor -- which is why that constructor takes the removed +// statements and the compile's own cause as arguments rather than deriving them. +// +// AND A COMPLETED CONTAINMENT COMPILE IS A REFUSAL HERE, WHICH IS THE OPPOSITE OF A FAILURE. The +// canary's receipt has exactly two blocking transitions and no spelling for "nothing blocked", so a +// stripped subject that COMPILES has no honest receipt on this carrier. Minting one anyway would be +// fabricating a blocker that did not occur; reporting it as its own refusal cause is how the day +// Step 0 actually succeeds becomes visible instead of being rendered as a stall. +// +// WHAT THIS PRODUCER DOES NOT ESTABLISH, stated because a receipt must not claim coverage it does +// not have. It observes the sources it was HANDED. That the handed vector is the whole of the tree +// the receipt names is the caller's obligation and is bound only by the canary's commit and tree +// pins; nothing here can check it. The next-rung trigger is a capability that COLLECTS the subject +// from the pinned tree, so the vector and the tree have one producer instead of two. + +type Step0StripProductionRefusalCause + = Step0SourceImportsNotParsed { + file: NonEmptyStr + cause: NonEmptyStr + } + | Step0SourceRewriteRefused { + file: NonEmptyStr + refusal: ImportStripRefusal + } + | Step0SubjectHadNoImportStatements + | Step0ContainmentCompileInstrumentRefused { + cause: NonEmptyStr + } + | Step0ContainmentCompileCompleted { + module_count: Int + } + +type Step0StripProduction + = Step0ReceiptProduced { + receipt: NamespaceStep0CanaryReceipt + } + | Step0StripProductionRefused { + cause: Step0StripProductionRefusalCause + } + +// One source, rewritten. The parse observation and the rewrite are two different authorities and +// both can refuse, so the two refusals stay apart: a parse that never ran and a span set the +// rewrite rejected have different owners and different repairs. +type Step0SourceStrip + = Step0SourceStripped { + source: FixtureSource + removed_statements: List + } + | Step0SourceStripRefused { + cause: Step0StripProductionRefusalCause + } + +fn namespace_step0_strip_source(source: FixtureSource) -> Step0SourceStrip { + match parsed_import_statements_live(file: source.path as String, source: source.content) { + ImportStatementParseRefused { cause: cause } => + Step0SourceStripRefused { cause: Step0SourceImportsNotParsed { file: source.path, cause: cause } } + ImportStatementsParsed { statements: statements } => + match strip_import_statements( + source_file: source.path as String, + source: source.content, + statements: statements + ) { + ImportStripRefused { refusal: refusal } => + Step0SourceStripRefused { cause: Step0SourceRewriteRefused { file: source.path, refusal: refusal } } + ImportStatementsStripped { rewritten_source: rewritten, removed_statements: removed } => + Step0SourceStripped { + source: FixtureSource { path: source.path, content: rewritten }, + removed_statements: removed + } + } + } +} + +// The absorbing arm is the refusal, and the first one is the one reported: a later source cannot +// repair an earlier one, and continuing would compile a subject already known to be unstripped. +type Step0StripFold + = Step0StripFolding { + sources: List + removed_statements: List + } + | Step0StripFoldHalted { + cause: Step0StripProductionRefusalCause + } + +fn namespace_step0_strip_fold_step(fold: Step0StripFold, source: FixtureSource) -> Step0StripFold { + match fold { + Step0StripFoldHalted { cause: cause } => Step0StripFoldHalted { cause: cause } + Step0StripFolding { sources: sources, removed_statements: removed } => + match namespace_step0_strip_source(source: source) { + Step0SourceStripRefused { cause: cause } => Step0StripFoldHalted { cause: cause } + Step0SourceStripped { source: stripped, removed_statements: source_removed } => + Step0StripFolding { + sources: sources |> list_push(stripped), + removed_statements: concat(removed, source_removed) + } + } + } +} + +fn namespace_step0_strip_subject(sources: List) -> Step0StripFold { + sources |> fold( + init: Step0StripFolding { sources: [], removed_statements: [] }, + f: (fold, source) => namespace_step0_strip_fold_step(fold: fold, source: source) + ) +} + +// The compile observation, folded to the one thing the canary's carrier can say about it: whether +// containment resolution refused, and with what cause. The instrument's three arms stay apart +// because a broken harness, a refusing subject and a passing subject have different owners: only +// the middle one is an observation of containment resolution at all. +type Step0ContainmentObservation + = Step0ContainmentResolutionRefused { + cause: NonEmptyStr + } + | Step0ContainmentObservationUnavailable { + cause: Step0StripProductionRefusalCause + } + +fn namespace_step0_containment_refusal_text(diagnostics: List) -> NonEmptyStr { + concat( + "containment resolution of the import-stripped subject refused with ", + concat(to_string(count(diagnostics |> filter(d => d.blocking))), " blocking diagnostic(s)") + ) as NonEmptyStr +} + +fn namespace_step0_observe_containment( + outcome: MultiModuleCompileFixtureOutcome +) -> Step0ContainmentObservation { + match outcome { + FixtureInstrumentRefused { cause: cause } => + Step0ContainmentObservationUnavailable { cause: Step0ContainmentCompileInstrumentRefused { cause: cause } } + FixtureCompileCompleted { module_count: module_count, emitted_files: _, diagnostics: _, source_digest: _, compiler_digest: _ } => + Step0ContainmentObservationUnavailable { cause: Step0ContainmentCompileCompleted { module_count: module_count } } + FixtureCompileRefused { module_count: _, diagnostics: diagnostics, source_digest: _, compiler_digest: _ } => + Step0ContainmentResolutionRefused { cause: namespace_step0_containment_refusal_text(diagnostics: diagnostics) } + } +} + +// THE FOLD, and the only caller the receipt constructor admits. Its arguments are the executed +// rewrite (the statements actually removed) and the observation that followed it (the containment +// compile's own cause); it invents neither. +// +// THE COMPILER IS ONE FACT AND ARRIVES FROM ONE CALLER. compiler_program travels beside +// compiler_executable_sha256 rather than being spelled here, because which program was run and +// which bytes it had are two halves of one observation about one dispatch; a literal here would +// make this module a second authority on the first half while the caller owned the second, and the +// two could then disagree with nothing to notice (review 58952). +fn namespace_step0_strip_receipt( + source_commit: GitObjectId, + source_tree: GitObjectId, + compiler_program: NonEmptyStr, + compiler_executable_sha256: Sha256Digest, + sources: List, + entry: NonEmptyStr +) -> Step0StripProduction { + match namespace_step0_strip_subject(sources: sources) { + Step0StripFoldHalted { cause: cause } => Step0StripProductionRefused { cause: cause } + Step0StripFolding { sources: stripped, removed_statements: removed } => + if count(removed) == 0 { + Step0StripProductionRefused { cause: Step0SubjectHadNoImportStatements } + } else { + match namespace_step0_observe_containment( + outcome: compile_fixture(fixture: MultiModuleCompileFixture { sources: stripped, entry: entry }) + ) { + Step0ContainmentObservationUnavailable { cause: cause } => + Step0StripProductionRefused { cause: cause } + Step0ContainmentResolutionRefused { cause: containment_cause } => + Step0ReceiptProduced { receipt: namespace_step0_strip_applied_receipt( + contract_identity: namespace_step0_canary_contract.identity, + source_commit: source_commit, + source_tree: source_tree, + compiler_program: compiler_program, + compiler_executable_sha256: compiler_executable_sha256, + input_subject_manifest: namespace_step0_manifest_identity(), + observed_tree: source_tree, + observed_import_statements: removed |> map(statement => statement as NonEmptyStr), + containment_compile_cause: containment_cause + ) } + } + } + } +} diff --git a/dag/gunbc/namespace/parsed_import_statement_bridge_seed_growth.dag b/dag/gunbc/namespace/parsed_import_statement_bridge_seed_growth.dag new file mode 100644 index 00000000000..59cdcff88d8 --- /dev/null +++ b/dag/gunbc/namespace/parsed_import_statement_bridge_seed_growth.dag @@ -0,0 +1,24 @@ +module gunbc.parsed_import_statement_bridge_seed_growth + +import std.decl_ref { DeclarationRef, WholeDeclaration } +import std.types { String } +import gunbc.roadmap_model { RoadmapNodeId } +import gunbc.seed_growth { SeedGrowthJustification } + +// Seed-growth obligation for the host bridge that carries the v1 parser's import-statement extents +// to the dag consumers that rewrite them. Homed beside the obligation it declares rather than in +// gunbc.seed_growth_admission, which owns the roster and the join. + +data parsed_import_statement_bridge_seed_growth_justification: SeedGrowthJustification = SeedGrowthJustification { + hand_authored_declarations: [ + DeclarationRef { + module_path: "v1_interpreter", + decl_name: "parsed_import_statements_value", + field: WholeDeclaration + } + ], + reason: "WHY A HOST BRIDGE RATHER THAN A .dag CONSUMER CALLING THE PRODUCER DIRECTLY, and it is the same reachability limit gunbc.namespace_structural_observation_bridge_seed_growth records. The statements are delimited by v1.compiler.parse parse_import_statement_extents -- the parser reading its own token consumption -- and the required floor resolves --source-root dag --source-root src/v2 only. Measured over the whole corpus: NO dag module imports a v1 module, so a consumer under dag cannot name that producer at all.\n\nWHAT THE BRIDGE IS AND IS NOT. It is a TRANSPORT: one in-memory source in, the parse's own observation out, encoded into std.import's carrier. It reads no repository file, needs no module index, and is hermetic by construction. It decides nothing -- where a statement starts and stops is decided by the parser, and what to do with that region is decided by std.import strip_import_statements. The single hand declaration is the encoder, and it encodes a coproduct rather than a list precisely so that 'the parse refused' and 'this module has no imports' cannot become one value at the boundary.\n\nWHY IT IS ADMITTED AGAINST THE v1 FREEZE: gunbc.v1_maintenance_standing v1_seed_standing admits by PURPOSE -- does the change serve the v2 self-host program. The import-deletion program is that program's current frontier, and gunbc.namespace_step0_canary records that Step 0 is blocked at ApplyTypedImportStrip for want of exactly this capability.\n\nHAND-ITEM DELTA: +1, the encoder named above, a free function in src/v1/stage0/src/v1_interpreter.rs and therefore citable as WholeDeclaration with no impl-method class. The dispatch arm is a new match arm inside the existing eval_builtin_inner macro, an ExistingSeedItemModified rather than a new declaration, and v1_interpreter_dispatch_generated.rs is generated from gunbc.v1_interpreter_primitive_surface. src/v1/02_parse.dag, dag/std/import.dag and src/v1/gunbc/parsed_import_statements.dag are substrate changes whose Rust mirrors are emitted.", + owning_dissolution_lane: "namespace-structural-observations" as RoadmapNodeId, + trigger: "Delete it when a dag consumer can obtain one in-memory source's parsed import statements WITHOUT a host bridge -- that is, when the parse that delimits them is reachable from a consumer whose resolution roots include the compiler that produces it, and the delimited statements for one source are obtainable there. What does NOT dissolve it is the required floor gaining src/v1 as a source root on its own: that is an artifact and can land while the producer is still unreachable from any executing consumer. The capability this bridge supplies is ONE IN-MEMORY SOURCE'S IMPORT STATEMENTS DELIMITED BY THE ORDINARY PARSE INSIDE AN EXECUTING CONSUMER, and the trigger is satisfied only when something else supplies exactly that.", + current_boundary: "src/v1/stage0/src/v1_interpreter.rs free_call.parsed_import_statements arm and its encoder; dag/gunbc/namespace/parsed_import_statement_production.dag; dag/gunbc/v1/v1_interpreter_primitive_surface.dag roster row; src/v1/04_method.dag builtin signature; dag/std/primitives.dag spelling roster" +} diff --git a/dag/gunbc/namespace/parsed_import_statement_production.dag b/dag/gunbc/namespace/parsed_import_statement_production.dag new file mode 100644 index 00000000000..9b2a8af27ba --- /dev/null +++ b/dag/gunbc/namespace/parsed_import_statement_production.dag @@ -0,0 +1,19 @@ +module gunbc.parsed_import_statement_production + +import std.import { ParsedImportStatements } +import std.types { String } + +// THE PRODUCTION SEAM FOR PARSED IMPORT STATEMENTS, AND IT IS A TRANSPORT RATHER THAN AN AUTHORITY. +// +// The statements are delimited by v1.compiler.parse parse_import_statement_extents -- the parser +// reading its own consumption -- and carried in std.import's vocabulary. No module under dag +// imports a v1 module anywhere in the corpus, and the required floor resolves `dag` and `src/v2` +// only, so a dag consumer cannot name that producer at all; this row is the one crossing. +// +// IT DECIDES NOTHING. One in-memory source in, the parse's own observation out. It reads no +// repository file and consults no module index, so a witness on it is hermetic. If a policy +// question ever arises about WHICH sources to ask about, it belongs in the caller. + +fn parsed_import_statements_live(file: String, source: String) -> ParsedImportStatements { + parsed_import_statements(file: file, source: source) +} diff --git a/dag/gunbc/seed_growth_admission.dag b/dag/gunbc/seed_growth_admission.dag index 7027bda4154..ab20316d474 100644 --- a/dag/gunbc/seed_growth_admission.dag +++ b/dag/gunbc/seed_growth_admission.dag @@ -11,6 +11,7 @@ import gunbc.claim_batch_function_preflight_seed_growth { claim_batch_function_p import gunbc.declaration_index_seed_growth { declaration_index_seed_growth_justification } import gunbc.emitted_closure_compile_seed_growth { emitted_closure_compile_seed_growth_justification } import gunbc.namespace_structural_observation_bridge_seed_growth { namespace_structural_observation_bridge_seed_growth_justification } +import gunbc.parsed_import_statement_bridge_seed_growth { parsed_import_statement_bridge_seed_growth_justification } import gunbc.namespace_wave_admission { namespace_wave_admission_seed_growth_justification } import gunbc.parse_refusal_location_seed_growth { parse_refusal_location_seed_growth_justification } import gunbc.qualified_pipe_callee_seed_growth { qualified_pipe_callee_seed_growth_justification } @@ -215,6 +216,7 @@ fn seed_growth_justification_roster() -> List { declaration_index_seed_growth_justification, emitted_closure_compile_seed_growth_justification, namespace_structural_observation_bridge_seed_growth_justification, + parsed_import_statement_bridge_seed_growth_justification, namespace_wave_admission_seed_growth_justification, qualified_pipe_callee_seed_growth_justification, target_invocation_seed_growth_justification, diff --git a/dag/gunbc/v1/v1_interpreter_primitive_surface.dag b/dag/gunbc/v1/v1_interpreter_primitive_surface.dag index 434b60a0a7f..931d96c52f6 100644 --- a/dag/gunbc/v1/v1_interpreter_primitive_surface.dag +++ b/dag/gunbc/v1/v1_interpreter_primitive_surface.dag @@ -1154,6 +1154,15 @@ fn v1_interpreter_authored_roster_arms() -> List + } + | ImportStatementParseRefused { + cause: NonEmptyStr + } + +type ImportStripRefusal + = ImportSpanNamesAnotherFile { + span_file: FilePath + source_file: FilePath + } + | ImportSpanOutsideSource { + span: SourceSpan + source_length: Int + } + | ImportSpanIsEmpty { + span: SourceSpan + } + | ImportSpanDoesNotDelimitAnImportStatement { + span: SourceSpan + source_text: String + } + | ImportSpanDoesNotNameItsModule { + span: SourceSpan + imported_module: String + source_text: String + } + | ImportSpansOutOfAscendingOrder { + previous_end: Int + next_start: Int + } + +type ImportStripOutcome + = ImportStatementsStripped { + rewritten_source: String + removed_statements: List + } + | ImportStripRefused { + refusal: ImportStripRefusal + } + +// THE HALTED ARM IS ABSORBING AND THE FIRST REFUSAL IS THE ONE REPORTED. A later statement cannot +// repair an earlier disagreement, and continuing past one would rewrite a source whose span set was +// already known not to describe it. +type ImportStripProgress + = ImportStripScanning { + cursor: Int + kept_segments: List + removed_statements: List + } + | ImportStripHalted { + refusal: ImportStripRefusal + } + +fn import_statement_keyword() -> String { + "import" +} + +// THE BOUNDARY THE CHECK IS ABOUT. `starts_with` and `string_contains` are substring predicates: +// they cannot tell `import xy` from a statement importing module `x`, nor a declaration named +// `imported_thing` from the keyword. The span comes from the parser, so a mismatch is already a +// typed refusal rather than a silent cut -- but a check softer than the thing it is guarding is a +// check that would not go red on the case it exists for. These two predicates read the span's text +// POSITIONALLY: the keyword occupies its own token, and the module path is the exact run of bytes +// that follows it, delimited by a character that cannot continue an identifier or a dotted path. +// The delimiter set is an ALLOWLIST, so an unanticipated following byte refuses rather than +// widening -- the failure arm never says "close enough". +fn is_import_whitespace(c: String) -> Bool { + c == " " || c == "\t" || c == "\n" || c == "\r" +} + +fn is_import_module_boundary(c: String) -> Bool { + is_import_whitespace(c: c) || c == "{" +} + +fn statement_text_opens_with_the_import_keyword(statement_text: String) -> Bool { + let keyword_length = string_length(s: import_statement_keyword()) + starts_with(s: statement_text, prefix: import_statement_keyword()) + && string_length(s: statement_text) > keyword_length + && is_import_whitespace(c: char_at(s: statement_text, pos: keyword_length)) +} + +// The whitespace run between the keyword and the module path is walked rather than assumed to be +// one character: a source that separates them with two spaces or a newline is a legitimate import +// statement, and refusing it would be a false refusal wearing fail-closed clothing. +fn import_module_start(statement_text: String, at: Int, limit: Int) -> Int { + if at >= limit { + at + } else if is_import_whitespace(c: char_at(s: statement_text, pos: at)) { + import_module_start(statement_text: statement_text, at: at + 1, limit: limit) + } else { + at + } +} + +fn statement_text_names_module(statement_text: String, imported_module: String) -> Bool { + let text_length = string_length(s: statement_text) + let module_start = import_module_start( + statement_text: statement_text, + at: string_length(s: import_statement_keyword()), + limit: text_length + ) + let module_end = module_start + string_length(s: imported_module) + string_length(s: imported_module) > 0 + && module_end <= text_length + && substring(s: statement_text, start: module_start, end: module_end) == imported_module + && (module_end == text_length || is_import_module_boundary(c: char_at(s: statement_text, pos: module_end))) +} + +// Segments are accumulated and joined once. Concatenating onto a growing string per statement is +// the copied accumulator DESIGN's bare-minimum-cost rule refuses regardless of the realized n. +fn import_strip_step( + progress: ImportStripProgress, + source_file: FilePath, + source: String, + source_length: Int, + statement: ParsedImportStatement +) -> ImportStripProgress { + match progress { + ImportStripHalted { refusal: refusal } => ImportStripHalted { refusal: refusal } + ImportStripScanning { cursor: cursor, kept_segments: kept_segments, removed_statements: removed_statements } => + if !(statement.span.file == source_file) { + ImportStripHalted { refusal: ImportSpanNamesAnotherFile { span_file: statement.span.file, source_file: source_file } } + } else if statement.span.start < 0 || statement.span.end > source_length { + ImportStripHalted { refusal: ImportSpanOutsideSource { span: statement.span, source_length: source_length } } + } else if statement.span.end <= statement.span.start { + ImportStripHalted { refusal: ImportSpanIsEmpty { span: statement.span } } + } else if statement.span.start < cursor { + ImportStripHalted { refusal: ImportSpansOutOfAscendingOrder { previous_end: cursor, next_start: statement.span.start } } + } else { + let statement_text = substring(s: source, start: statement.span.start, end: statement.span.end) + if !statement_text_opens_with_the_import_keyword(statement_text: statement_text) { + ImportStripHalted { refusal: ImportSpanDoesNotDelimitAnImportStatement { span: statement.span, source_text: statement_text } } + } else if !statement_text_names_module(statement_text: statement_text, imported_module: statement.imported_module) { + ImportStripHalted { refusal: ImportSpanDoesNotNameItsModule { span: statement.span, imported_module: statement.imported_module, source_text: statement_text } } + } else { + ImportStripScanning { + cursor: statement.span.end, + kept_segments: kept_segments |> list_push(substring(s: source, start: cursor, end: statement.span.start)), + removed_statements: removed_statements |> list_push(statement_text) + } + } + } + } +} + +// THE ONE AUTHORITY. Removing several statements is not removing one several times: every removal +// shifts the offsets of the ones after it, so a per-statement operation applied repeatedly would +// have to re-derive spans it was not given. This fold consumes the ascending span set once and +// rebuilds the source from the gaps between them, which is why the singular case below is this +// operation on a one-element list rather than a second implementation beside it. +fn strip_import_statements( + source_file: FilePath, + source: String, + statements: List +) -> ImportStripOutcome { + let source_length = string_length(s: source) + let scanned = statements |> fold( + init: ImportStripScanning { cursor: 0, kept_segments: [], removed_statements: [] }, + f: (progress, statement) => import_strip_step( + progress: progress, + source_file: source_file, + source: source, + source_length: source_length, + statement: statement + ) + ) + match scanned { + ImportStripHalted { refusal: refusal } => ImportStripRefused { refusal: refusal } + ImportStripScanning { cursor: cursor, kept_segments: kept_segments, removed_statements: removed_statements } => + ImportStatementsStripped { + rewritten_source: join(kept_segments |> list_push(substring(s: source, start: cursor, end: source_length)), ""), + removed_statements: removed_statements + } + } +} + +fn strip_one_import_statement( + source_file: FilePath, + source: String, + statement: ParsedImportStatement +) -> ImportStripOutcome { + strip_import_statements(source_file: source_file, source: source, statements: [statement]) +} diff --git a/dag/std/primitives.dag b/dag/std/primitives.dag index 11299c81ad8..54654511c31 100644 --- a/dag/std/primitives.dag +++ b/dag/std/primitives.dag @@ -606,6 +606,7 @@ fn builtin_registry_surface_names() -> List { "doc_graph_dangling_link_count", "doc_graph_doc_count", "compile_dag_rust_emit_check", + "parsed_import_statements", "namespace_structural_observation_admissions", "witness_layer_roots_compile_clean_check", "witness_layer_roots_compile_clean_emit_check", diff --git a/dag/test/claim/namespace_step0_strip_producer_witness_test.dag b/dag/test/claim/namespace_step0_strip_producer_witness_test.dag new file mode 100644 index 00000000000..a6294c01ac1 --- /dev/null +++ b/dag/test/claim/namespace_step0_strip_producer_witness_test.dag @@ -0,0 +1,403 @@ +module test.claim.namespace_step0_strip_producer_witness + +import std.types { Bool, Int, List, NonEmptyStr, SourceSpan, String } +import std.content_hash { Sha256Digest, sha256_hex_digest } +import std.import { + ImportStatementsParsed, + ImportStatementParseRefused, + ImportStatementsStripped, + ImportStripRefused, + ImportSpanNamesAnotherFile, + ImportSpanDoesNotDelimitAnImportStatement, + ImportSpansOutOfAscendingOrder, + ParsedImportStatement, + ParsedImportStatements, + ImportStripOutcome, + ImportStripRefusal, + strip_import_statements, + strip_one_import_statement, +} +import extdeps.git.object_store { GitObjectId, git_sha1_object_id } +import v2.std.live_tree { LiveTreeDisposition, SubstrateInputsOnly } +import gunbc.parsed_import_statement_production { parsed_import_statements_live } +import tools.multi_module_compile_fixture { FixtureSource } +import gunbc.namespace_step0_canary { + NamespaceStep0CanaryReceipt, + NamespaceStep0CanaryAdmission, + NamespaceStep0ReceiptAdmitted, + NamespaceStep0ReceiptRefused, + NoStep0Observation, + Step0StripObservation, + admit_namespace_step0_canary_receipt, +} +import gunbc.namespace_step0_strip_producer { + Step0StripProduction, + Step0ReceiptProduced, + Step0StripProductionRefused, + Step0StripProductionRefusalCause, + Step0SubjectHadNoImportStatements, + Step0ContainmentCompileCompleted, + Step0SourceImportsNotParsed, + namespace_step0_strip_receipt, +} + +// EXECUTED EVIDENCE FOR THE TWO CAPABILITIES gunbc.namespace_step0_canary names as missing at +// ApplyTypedImportStrip: a typed parse-level rewrite that removes one complete import statement, +// and an owned producer that applies it and folds the compile observation that follows. +// +// THIS ENTRY DOES NOT CLAIM STEP 0 MOVED. It does not touch the frozen contract, the classifier, +// the subject-entry rule or the controls in test.claim.namespace_step0_canary_witness, and it mints +// no receipt over the repository tree. What it proves is that a receipt at the longer frontier is +// now PRODUCIBLE by execution rather than only authorable by a fixture: the producer strips a real +// subject, compiles it, and the frozen classifier admits what came back. Rebinding the canary's +// baseline to a corpus-scale run is a separate change, exactly as the freeze note requires. +// +// HERMETIC: every subject below is an in-memory source. The parse bridge reads no repository file +// and the multi-module compile instrument has no corpus pool, which is what makes the stripped +// subjects refuse for the reason the witness is about rather than by pool-membership coincidence. + +data live_tree_disposition: LiveTreeDisposition = SubstrateInputsOnly + +fn provider_path() -> NonEmptyStr { "w/provider.dag" as NonEmptyStr } +fn entry_path() -> NonEmptyStr { "w/entry.dag" as NonEmptyStr } + +fn provider_source() -> String { + "module w.provider\n\nimport std.types { NonEmptyStr }\n\nfn w_provided() -> NonEmptyStr {\n \"x\" as NonEmptyStr\n}\n" +} + +// A braced member list broken over three lines: the case a line-oriented filter gets wrong, and the +// reason the rewrite takes a parser-delimited span rather than searching the text. +fn entry_source() -> String { + "module w.entry\n\nimport w.provider {\n w_provided,\n}\nimport std.types { NonEmptyStr }\n\nfn w_entry() -> NonEmptyStr {\n w_provided()\n}\n" +} + +// MEASURED, NOT ASSUMED: with both imports removed this pair STILL COMPILES. `Bool` is a kernel +// type that needs no import, and `w_provided` is reached by bare-name resolution across the two +// supplied modules, so the imports were carrying nothing the resolution could not recover. It is the +// same "compiles only by pool-membership coincidence" shape v2.lens.module_graph records from #7908, +// reproduced here as the subject that proves the producer refuses instead of inventing a blocker. +fn kernel_typed_provider_source() -> String { + "module w.kprovider\n\nimport std.types { Bool }\n\nfn w_kprovided() -> Bool {\n true\n}\n" +} + +fn kernel_typed_entry_source() -> String { + "module w.kentry\n\nimport w.kprovider { w_kprovided }\nimport std.types { Bool }\n\nfn w_kentry() -> Bool {\n w_kprovided()\n}\n" +} + +fn importless_source() -> String { + "module w.alone\n\ntype WAlone = WAloneOnly\n\nfn w_alone() -> WAlone {\n WAloneOnly\n}\n" +} + +fn subject_sources() -> List { + [ + FixtureSource { path: provider_path(), content: provider_source() }, + FixtureSource { path: entry_path(), content: entry_source() }, + ] +} + +fn statements_of(file: NonEmptyStr, source: String) -> List { + match parsed_import_statements_live(file: file as String, source: source) { + ImportStatementParseRefused { cause: _ } => [] + ImportStatementsParsed { statements: statements } => statements + } +} + +fn stripped_source_of(file: NonEmptyStr, source: String) -> String { + match strip_import_statements( + source_file: file as String, + source: source, + statements: statements_of(file: file, source: source) + ) { + ImportStripRefused { refusal: _ } => "" + ImportStatementsStripped { rewritten_source: rewritten, removed_statements: _ } => rewritten + } +} + +// ------------------------------------------------------------------ the parse observation + +test fn the_parser_delimits_both_import_statements_of_the_entry_source() -> Bool { + count(statements_of(file: entry_path(), source: entry_source())) == 2 +} + +// A source that is not a module has no import statements to delimit, and says so rather than +// reporting none -- the distinction the carrier is a coproduct for. +test fn a_source_that_is_not_a_module_refuses_rather_than_reporting_no_imports() -> Bool { + match parsed_import_statements_live(file: "w/none.dag", source: "fn w_loose() -> Bool { true }\n") { + ImportStatementsParsed { statements: _ } => false + ImportStatementParseRefused { cause: _ } => true + } +} + +// ------------------------------------------------------------------ the typed rewrite + +// The multi-line braced statement is removed WHOLE, and the statement after it survives its own +// removal at the shifted offsets -- the property a per-statement rewrite applied repeatedly cannot +// have, and the reason the fold consumes the ascending span set once. +test fn the_rewrite_removes_every_complete_import_statement_of_the_entry_source() -> Bool { + stripped_source_of(file: entry_path(), source: entry_source()) + == "module w.entry\n\n\n\n\nfn w_entry() -> NonEmptyStr {\n w_provided()\n}\n" +} + +test fn the_rewrite_reports_the_statements_it_removed() -> Bool { + match strip_import_statements( + source_file: entry_path() as String, + source: entry_source(), + statements: statements_of(file: entry_path(), source: entry_source()) + ) { + ImportStripRefused { refusal: _ } => false + ImportStatementsStripped { rewritten_source: _, removed_statements: removed } => + removed == ["import w.provider {\n w_provided,\n}", "import std.types { NonEmptyStr }"] + } +} + +// DISCRIMINATING RED. A span taken from one file cannot cut a hole in another: the caller holds the +// source and the parse holds the span, and nothing else would notice the mismatch. +test fn a_span_from_another_file_is_refused_rather_than_applied() -> Bool { + match strip_one_import_statement( + source_file: entry_path() as String, + source: entry_source(), + statement: ParsedImportStatement { + span: SourceSpan { file: provider_path() as String, start: 12, end: 37 }, + imported_module: "std.types" + } + ) { + ImportStatementsStripped { rewritten_source: _, removed_statements: _ } => false + ImportStripRefused { refusal: refusal } => + match refusal { + ImportSpanNamesAnotherFile { span_file: _, source_file: _ } => true + _ => false + } + } +} + +// DISCRIMINATING RED. A well-formed span over bytes that are not an import statement is refused, +// which is what makes the operation parse-level rather than an offset-driven text cut. +test fn a_span_that_does_not_delimit_an_import_statement_is_refused() -> Bool { + match strip_one_import_statement( + source_file: entry_path() as String, + source: entry_source(), + statement: ParsedImportStatement { + span: SourceSpan { file: entry_path() as String, start: 0, end: 10 }, + imported_module: "w.entry" + } + ) { + ImportStatementsStripped { rewritten_source: _, removed_statements: _ } => false + ImportStripRefused { refusal: refusal } => + match refusal { + ImportSpanDoesNotDelimitAnImportStatement { span: _, source_text: _ } => true + _ => false + } + } +} + +// DISCRIMINATING RED, and the one review 58968 asked for. `w.provid` is a proper PREFIX of the +// module the span actually imports, so a `string_contains` check would have said the span names it +// and cut the statement out under a module path that does not exist. The boundary is positional: +// the module path is the exact run after the keyword, delimited by a character that cannot continue +// it, and `e` can. +test fn a_span_naming_a_prefix_of_the_module_it_imports_is_refused() -> Bool { + match strip_one_import_statement( + source_file: entry_path() as String, + source: entry_source(), + statement: ParsedImportStatement { + span: SourceSpan { file: entry_path() as String, start: 16, end: 51 }, + imported_module: "w.provid" + } + ) { + ImportStatementsStripped { rewritten_source: _, removed_statements: _ } => false + ImportStripRefused { refusal: refusal } => + match refusal { + ImportSpanDoesNotNameItsModule { span: _, imported_module: _, source_text: _ } => true + _ => false + } + } +} + +// DISCRIMINATING RED, the same softness read from the other end: a declaration whose name merely +// BEGINS with the keyword is not an import statement. `starts_with` alone admitted this text. +test fn a_span_whose_text_only_begins_with_the_keyword_is_refused() -> Bool { + match strip_one_import_statement( + source_file: entry_path() as String, + source: "import_alias w.provider\n", + statement: ParsedImportStatement { + span: SourceSpan { file: entry_path() as String, start: 0, end: 23 }, + imported_module: "w.provider" + } + ) { + ImportStatementsStripped { rewritten_source: _, removed_statements: _ } => false + ImportStripRefused { refusal: refusal } => + match refusal { + ImportSpanDoesNotDelimitAnImportStatement { span: _, source_text: _ } => true + _ => false + } + } +} + +// DISCRIMINATING RED. Descending spans are refused rather than silently rebuilding the source from +// gaps that overlap, which would drop or duplicate bytes with no diagnostic. +test fn descending_spans_are_refused_rather_than_rebuilding_the_source() -> Bool { + match strip_import_statements( + source_file: entry_path() as String, + source: entry_source(), + statements: reverse(statements_of(file: entry_path(), source: entry_source())) + ) { + ImportStatementsStripped { rewritten_source: _, removed_statements: _ } => false + ImportStripRefused { refusal: refusal } => + match refusal { + ImportSpansOutOfAscendingOrder { previous_end: _, next_start: _ } => true + _ => false + } + } +} + +// ------------------------------------------------------------------ the owned producer + +fn witness_commit() -> GitObjectId? { + git_sha1_object_id(hex: "1111111111111111111111111111111111111111") +} + +fn witness_tree() -> GitObjectId? { + git_sha1_object_id(hex: "2222222222222222222222222222222222222222") +} + +fn witness_compiler() -> Sha256Digest? { + sha256_hex_digest(hex: "3333333333333333333333333333333333333333333333333333333333333333") +} + +fn production_over(sources: List, entry: NonEmptyStr) -> Step0StripProduction? { + match witness_commit() { + Absent => none + Present { value: commit } => match witness_tree() { + Absent => none + Present { value: tree } => match witness_compiler() { + Absent => none + Present { value: compiler } => Present { value: namespace_step0_strip_receipt( + source_commit: commit, + source_tree: tree, + compiler_program: "target/release/gunbc" as NonEmptyStr, + compiler_executable_sha256: compiler, + sources: sources, + entry: entry + ) } + } + } + } +} + +fn admitted(receipt: NamespaceStep0CanaryReceipt) -> Bool { + match witness_commit() { + Absent => false + Present { value: commit } => match witness_tree() { + Absent => false + Present { value: tree } => match witness_compiler() { + Absent => false + Present { value: compiler } => + match admit_namespace_step0_canary_receipt( + receipt: receipt, + expected_commit: commit, + expected_tree: tree, + expected_compiler_sha256: compiler + ) { + NamespaceStep0ReceiptAdmitted => true + NamespaceStep0ReceiptRefused { field: _, cause: _ } => false + } + } + } + } +} + +fn production_refused_with(production: Step0StripProduction, wanted: Step0StripProductionRefusalCause) -> Bool { + match production { + Step0ReceiptProduced { receipt: _ } => false + Step0StripProductionRefused { cause: cause } => cause == wanted + } +} + +// THE MOVEMENT CLAIM, green by execution and not by authorship: the producer strips a real subject, +// compiles what it produced, and the FROZEN classifier admits the receipt at the strictly longer +// frontier. Nothing in this test constructs a receipt. +test fn the_producer_mints_a_receipt_the_frozen_classifier_admits() -> Bool { + match production_over(sources: subject_sources(), entry: entry_path()) { + Absent => false + Present { value: production } => + match production { + Step0StripProductionRefused { cause: _ } => false + Step0ReceiptProduced { receipt: receipt } => admitted(receipt: receipt) + } + } +} + +test fn the_produced_receipt_carries_the_statements_the_rewrite_removed() -> Bool { + match production_over(sources: subject_sources(), entry: entry_path()) { + Absent => false + Present { value: production } => + match production { + Step0StripProductionRefused { cause: _ } => false + Step0ReceiptProduced { receipt: receipt } => + match receipt.observed_subject { + NoStep0Observation => false + Step0StripObservation { observed_tree: _, observed_manifest: _, observed_import_statements: observed } => + count(observed) == 3 + } + } + } +} + +// DISCRIMINATING RED. A subject with no import statements did not enter the Step 0 subject, and the +// producer says so rather than minting a receipt whose observation is empty -- the shape the frozen +// classifier would refuse at ObservedSubjectField, reached here one stage earlier. +test fn a_subject_with_no_import_statements_refuses_rather_than_claiming_entry() -> Bool { + match production_over( + sources: [FixtureSource { path: "w/alone.dag" as NonEmptyStr, content: importless_source() }], + entry: "w/alone.dag" as NonEmptyStr + ) { + Absent => false + Present { value: production } => + production_refused_with(production: production, wanted: Step0SubjectHadNoImportStatements) + } +} + +// DISCRIMINATING RED, and the arm that matters most. A stripped subject that COMPILES has no +// blocking transition, and the canary's receipt has no spelling for one. The producer refuses with +// the compile's own module count rather than fabricating a blocker that did not occur. +test fn a_stripped_subject_that_still_compiles_refuses_rather_than_fabricating_a_blocker() -> Bool { + match production_over( + sources: [ + FixtureSource { path: "w/kprovider.dag" as NonEmptyStr, content: kernel_typed_provider_source() }, + FixtureSource { path: "w/kentry.dag" as NonEmptyStr, content: kernel_typed_entry_source() }, + ], + entry: "w/kentry.dag" as NonEmptyStr + ) { + Absent => false + Present { value: production } => + match production { + Step0ReceiptProduced { receipt: _ } => false + Step0StripProductionRefused { cause: cause } => + match cause { + Step0ContainmentCompileCompleted { module_count: _ } => true + _ => false + } + } + } +} + +// DISCRIMINATING RED. A source whose imports could not be delimited is not a source with none: the +// producer refuses at the parse rather than reporting a clean rewrite of a file it never read. +test fn a_subject_whose_imports_did_not_parse_refuses_at_the_parse() -> Bool { + match production_over( + sources: [FixtureSource { path: "w/loose.dag" as NonEmptyStr, content: "fn w_loose() -> Bool { true }\n" }], + entry: "w/loose.dag" as NonEmptyStr + ) { + Absent => false + Present { value: production } => + match production { + Step0ReceiptProduced { receipt: _ } => false + Step0StripProductionRefused { cause: cause } => + match cause { + Step0SourceImportsNotParsed { file: _, cause: _ } => true + _ => false + } + } + } +} + diff --git a/src/v1/02_parse.dag b/src/v1/02_parse.dag index e9643f650d7..1b5e20929d4 100644 --- a/src/v1/02_parse.dag +++ b/src/v1/02_parse.dag @@ -41,7 +41,7 @@ import v1.std.core { ShLitStr, ShLitInt, ShLitFloat, ShIdent, ShCaret, ShStrBegin, ShStrMid, ShStrEnd, ShNewline, ShEof, ShUnknown, - SourceSpan, + SourceSpan, NonEmptyStr, ErrorNode, make_error_node, ParseError, InternalError, with_required_cardinality, error_type, is_compiler_error, @@ -81,6 +81,12 @@ import std.occurrence_identity { node_occurrence_identity_minted, } import std.algebra { FreeMonoid, Empty, Cons } +import std.import { + ParsedImportStatement, + ParsedImportStatements, + ImportStatementsParsed, + ImportStatementParseRefused, +} import std.syntax { Add, BinOp, LitStr, LiteralValue, @@ -1758,22 +1764,24 @@ fn occurrence_transport_from_parse_context(ctx: ParseContext) -> OccurrenceTrans } } -fn parse_with_table_at( +// The opening parse context for one token vector. Extracted so the import-extent reading below +// enters the grammar under the SAME context the full reading does rather than assembling a second +// one beside it -- a context that differed in interning or occurrence allocation would be a second +// answer to what the parser sees. +fn parse_context_for_tokens( tokens: List, source_indices: Map, intern_table: InternTable, occurrence_base: AuthoredTokenOrdinalSpace, heads_only: Bool -) -> ParseWithTableResult { - let occurrence_allocator = occurrence_id_allocator_advance_to( - alloc: occurrence_base.allocator, - min_next: intern_table.authored_token_ordinals - ) - let pre_interned = pre_intern_tokens(tokens: tokens, table: intern_table) - let ctx = ParseContext { +) -> ParseContext { + ParseContext { source_indices: source_indices, - intern_table: pre_interned, - occurrence_allocator: occurrence_allocator, + intern_table: pre_intern_tokens(tokens: tokens, table: intern_table), + occurrence_allocator: occurrence_id_allocator_advance_to( + alloc: occurrence_base.allocator, + min_next: intern_table.authored_token_ordinals + ), occurrence_index: Present { value: OccurrenceIndex { entries: [] }, }, @@ -1781,6 +1789,26 @@ fn parse_with_table_at( reference_occurrences: Present { value: [] }, heads_only: heads_only, } +} + +fn parse_with_table_at( + tokens: List, + source_indices: Map, + intern_table: InternTable, + occurrence_base: AuthoredTokenOrdinalSpace, + heads_only: Bool +) -> ParseWithTableResult { + let occurrence_allocator = occurrence_id_allocator_advance_to( + alloc: occurrence_base.allocator, + min_next: intern_table.authored_token_ordinals + ) + let ctx = parse_context_for_tokens( + tokens: tokens, + source_indices: source_indices, + intern_table: intern_table, + occurrence_base: occurrence_base, + heads_only: heads_only + ) let r = parse_module(tokens: token_stream_new(all: tokens), ctx: ctx) if has_err(err: r.err) { let occurrence_transport = occurrence_transport_from_parse_context(ctx: r.ctx) @@ -1960,6 +1988,97 @@ fn parse_imports_acc(tokens: TokenStream, ctx: ParseContext, acc: List) -> } } +// THE EXTENT OF AN IMPORT STATEMENT IS A FACT ABOUT ITS PARSE, and this is the only production +// that can state it. An import node carries the `import` keyword as its span and the module path as +// its ident_span, so where the statement STOPS is known only while its tokens are being consumed -- +// and a consumer that needs to remove one (std.import) must be handed a delimited region rather +// than left to recognise `import` in the text, which a member list running over several lines and a +// string containing the word both defeat. +// +// THERE IS NO SECOND GRAMMAR HERE. Statements are produced by parse_import, exactly as the full +// reading produces them; this walk decides only where that production started and stopped, from the +// token stream's own position before and after it. A rule added or corrected in parse_import is +// reflected here with no edit. +// The last non-newline token in a consumed window. parse_import consumes the newlines that FOLLOW a +// statement and they are not part of it: ending the extent at them would make removing a statement +// also remove the blank line after it, which is a second and unstated edit. +fn last_consumed_token_end(all: List, from: Int, until: Int, end: Int?) -> Int? { + if from >= until { + end + } else { + match get(xs: all, index: from) { + Absent => last_consumed_token_end(all: all, from: from + 1, until: until, end: end) + Present { value: t } => + if is_newline_shape(shape: t.shape) { + last_consumed_token_end(all: all, from: from + 1, until: until, end: end) + } else { + last_consumed_token_end(all: all, from: from + 1, until: until, end: Present { value: t.span.end }) + } + } + } +} + +fn parse_import_statement_extents_acc( + tokens: TokenStream, + ctx: ParseContext, + acc: List +) -> ParsedImportStatements { + let tokens = skip_newlines(tokens: tokens) + if tok_is_keyword(tok: token_stream_first(stream: tokens), kw: "import") { + let before = tokens.pos + let r = parse_import(tokens: tokens, ctx: ctx) + if has_err(err: r.err) { + return ImportStatementParseRefused { cause: "an import statement did not parse" as NonEmptyStr } + } + match token_stream_first(stream: tokens) { + Absent => ImportStatementParseRefused { cause: "an import statement parsed from no token" as NonEmptyStr } + Present { value: first } => + match last_consumed_token_end(all: tokens.all, from: before, until: r.tokens.pos, end: none) { + Absent => ImportStatementParseRefused { cause: "an import statement parsed without consuming a token" as NonEmptyStr } + Present { value: end } => + parse_import_statement_extents_acc( + tokens: r.tokens, + ctx: parse_context_after_node(ctx: r.ctx, node: r.import), + acc: list_push(acc, ParsedImportStatement { + span: SourceSpan { file: first.span.file, start: first.span.start, end: end }, + imported_module: r.import.name + }) + ) + } + } + } else { + ImportStatementsParsed { statements: acc } + } +} + +// The reading is entered at the module declaration because that is where the grammar admits +// imports; reaching them costs one `expect` and one parse_dotted_ident, both parser primitives and +// neither a production of its own. A source that does not open with a module declaration has no +// import statements to delimit, and says so rather than reporting none. +fn parse_import_statement_extents( + tokens: List, + source_indices: Map, + intern_table: InternTable +) -> ParsedImportStatements { + let ctx = parse_context_for_tokens( + tokens: tokens, + source_indices: source_indices, + intern_table: intern_table, + occurrence_base: intern_table.authored_token_ordinals, + heads_only: false + ) + let stream = skip_newlines(tokens: token_stream_new(all: tokens)) + let r = expect(tokens: stream, expected: ExpectKeyword { text: "module" }) + if has_err(err: r.err) { + return ImportStatementParseRefused { cause: "source does not open with a module declaration" as NonEmptyStr } + } + let r = parse_dotted_ident(tokens: r.tokens) + if has_err(err: r.err) { + return ImportStatementParseRefused { cause: "module declaration does not name a dotted module path" as NonEmptyStr } + } + parse_import_statement_extents_acc(tokens: skip_newlines(tokens: r.tokens), ctx: ctx, acc: []) +} + fn parse_items(tokens: TokenStream, ctx: ParseContext) -> ItemsResult { parse_items_acc(tokens: tokens, ctx: ctx, acc: []) } diff --git a/src/v1/04_method.dag b/src/v1/04_method.dag index d4a0455c218..3444b7c7ea3 100644 --- a/src/v1/04_method.dag +++ b/src/v1/04_method.dag @@ -344,6 +344,7 @@ data builtin_function_registry: Map = { "compile_dag_rust_emit_check": BuiltinSignature { params: [BuiltinParam { name: "source", ty: NamedTemplate { name: "String" } }, BuiltinParam { name: "entry", ty: NamedTemplate { name: "String" } }, BuiltinParam { name: "expected", ty: NamedTemplate { name: "String" } }, BuiltinParam { name: "mode", ty: NamedTemplate { name: "String" } }], returns: bool_type }, "compile_dag_diagnostic_census": BuiltinSignature { params: [BuiltinParam { name: "source", ty: NamedTemplate { name: "String" } }], returns: type_variable_node(id: "compile_diagnostic_census_result") }, "compile_dag_multi_module_fixture": BuiltinSignature { params: [BuiltinParam { name: "sources", ty: ContainerOf { source: Named { name: "List" }, element: NamedTemplate { name: "String" } } }, BuiltinParam { name: "entry", ty: NamedTemplate { name: "String" } }, BuiltinParam { name: "function", ty: NamedTemplate { name: "String" } }], returns: type_variable_node(id: "multi_module_compile_fixture_result") }, + "parsed_import_statements": BuiltinSignature { params: [BuiltinParam { name: "file", ty: NamedTemplate { name: "String" } }, BuiltinParam { name: "source", ty: NamedTemplate { name: "String" } }], returns: type_variable_node(id: "parsed_import_statements_result") }, "observe_declared_import_closure_symbol_binding": BuiltinSignature { params: [BuiltinParam { name: "pool_roots", ty: ContainerOf { source: Named { name: "List" }, element: NamedTemplate { name: "String" } } }, BuiltinParam { name: "entry_path", ty: NamedTemplate { name: "String" } }, BuiltinParam { name: "consumer_module", ty: NamedTemplate { name: "String" } }, BuiltinParam { name: "symbol", ty: NamedTemplate { name: "String" } }], returns: type_variable_node(id: "declared_import_closure_binding_result") }, "class_b_import_closure_gate_not_affected_skip": BuiltinSignature { params: [], returns: bool_type }, "witness_layer_roots_compile_clean_check": BuiltinSignature { params: [], returns: bool_type }, diff --git a/src/v1/gunbc/parsed_import_statements.dag b/src/v1/gunbc/parsed_import_statements.dag new file mode 100644 index 00000000000..cd6be8ab211 --- /dev/null +++ b/src/v1/gunbc/parsed_import_statements.dag @@ -0,0 +1,21 @@ +module v1.gunbc.parsed_import_statements + +import std.types { String } +import std.import { ParsedImportStatements } +import v1.compiler.tokenize { tokenize } +import v1.compiler.parse { parse_import_statement_extents } +import v1.std.core { build_newline_index, empty_intern_table, NewlineIndex } + +// THE TRANSPORT FROM ONE IN-MEMORY SOURCE TO ITS PARSED IMPORT STATEMENTS, and nothing else. +// It reads no repository file, consults no module index and decides nothing: the statements are +// delimited by v1.compiler.parse parse_import_statement_extents, which is the parser reading its +// own consumption, and the carrier is std.import's. It exists because a caller outside the seed +// holds a source string and cannot assemble a token vector, a newline index and an intern table. + +fn parsed_import_statements(file: String, source: String) -> ParsedImportStatements { + parse_import_statement_extents( + tokens: tokenize(source: source, file: file), + source_indices: map_insert(empty_map(), file, build_newline_index(file: file, source: source)), + intern_table: empty_intern_table(), + ) +} diff --git a/src/v1/stage0/src/emitted_population.rs b/src/v1/stage0/src/emitted_population.rs index b61421cc313..4cdc709b50f 100644 --- a/src/v1/stage0/src/emitted_population.rs +++ b/src/v1/stage0/src/emitted_population.rs @@ -55,6 +55,7 @@ // src/std_error_primitives.rs // src/std_execution_mode.rs // src/std_graph.rs +// src/std_import.rs // src/std_induction.rs // src/std_integer.rs // src/std_interface_summary.rs @@ -138,6 +139,7 @@ // src/v1_compiler_workspace_members.rs // src/v1_gunbc_namespace_reference_derived_closure_production_observations.rs // src/v1_gunbc_occurrence_binding_parser_walk.rs +// src/v1_gunbc_parsed_import_statements.rs // src/v1_probe_emit_interp.rs // src/v1_rt.rs // src/v1_std_core.rs diff --git a/src/v1/stage0/src/lib.rs b/src/v1/stage0/src/lib.rs index f3ad199b634..bf6e63e487b 100644 --- a/src/v1/stage0/src/lib.rs +++ b/src/v1/stage0/src/lib.rs @@ -191,6 +191,17 @@ pub mod std_constructors; suspicious_double_ref_op, clippy::all )] +pub mod std_import; +#[allow( + unused_imports, + unused_variables, + unused_mut, + unused_parens, + dead_code, + non_shorthand_field_patterns, + suspicious_double_ref_op, + clippy::all +)] pub mod std_integer; #[allow( unused_imports, @@ -708,6 +719,17 @@ pub mod v1_gunbc_occurrence_binding_parser_walk; suspicious_double_ref_op, clippy::all )] +pub mod v1_gunbc_parsed_import_statements; +#[allow( + unused_imports, + unused_variables, + unused_mut, + unused_parens, + dead_code, + non_shorthand_field_patterns, + suspicious_double_ref_op, + clippy::all +)] pub mod v1_probe_emit_interp; #[allow( unused_imports, diff --git a/src/v1/stage0/src/std_import.rs b/src/v1/stage0/src/std_import.rs new file mode 100644 index 00000000000..61a21efcbbe --- /dev/null +++ b/src/v1/stage0/src/std_import.rs @@ -0,0 +1,311 @@ +// Generated by v1 compiler -- do not edit. +// Source module: std.import + +use self::ImportStripOutcome::*; +use self::ImportStripProgress::*; +use self::ImportStripRefusal::*; +use self::ParsedImportStatements::*; +pub use crate::std_types::{FilePath, List, NonEmptyStr, SourceSpan}; +use crate::v1_rt; +use crate::v1_rt::{VecCompat, VecJoin}; +use crate::NonEmptyBTreeSet; +use crate::NonEmptyVec; +use im::{vector as vec, HashMap, OrdSet as BTreeSet, Vector as Vec}; +use std::rc::Rc; + +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +pub struct ParsedImportStatement { + pub span: Rc, + pub imported_module: String, +} + +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +#[serde(tag = "_variant")] +pub enum ParsedImportStatements { + ImportStatementsParsed { + statements: Rc>>, + }, + ImportStatementParseRefused { + cause: NonEmptyStr, + }, +} + +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +#[serde(tag = "_variant")] +pub enum ImportStripRefusal { + ImportSpanNamesAnotherFile { + span_file: FilePath, + source_file: FilePath, + }, + ImportSpanOutsideSource { + span: Rc, + source_length: i64, + }, + ImportSpanIsEmpty { + span: Rc, + }, + ImportSpanDoesNotDelimitAnImportStatement { + span: Rc, + source_text: String, + }, + ImportSpanDoesNotNameItsModule { + span: Rc, + imported_module: String, + source_text: String, + }, + ImportSpansOutOfAscendingOrder { + previous_end: i64, + next_start: i64, + }, +} + +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +#[serde(tag = "_variant")] +pub enum ImportStripOutcome { + ImportStatementsStripped { + rewritten_source: String, + removed_statements: Rc>, + }, + ImportStripRefused { + refusal: Rc, + }, +} + +#[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] +#[serde(tag = "_variant")] +pub enum ImportStripProgress { + ImportStripScanning { + cursor: i64, + kept_segments: Rc>, + removed_statements: Rc>, + }, + ImportStripHalted { + refusal: Rc, + }, +} + +pub fn import_statement_keyword() -> String { + "import".to_string() +} + +pub fn is_import_whitespace(c: String) -> bool { + ((((c.clone() == " ".to_string()) || (c.clone() == "\t".to_string())) + || (c.clone() == "\n".to_string())) + || (c.clone() == "\x0d".to_string())) +} + +pub fn is_import_module_boundary(c: String) -> bool { + (is_import_whitespace(c.clone()) || (c.clone() == "{".to_string())) +} + +pub fn statement_text_opens_with_the_import_keyword(statement_text: String) -> bool { + { + let keyword_length = v1_rt::string_length(&import_statement_keyword()); + ((v1_rt::starts_with(statement_text.clone(), import_statement_keyword()) + && (v1_rt::string_length(&statement_text) > keyword_length.clone())) + && is_import_whitespace(v1_rt::char_at(&statement_text, keyword_length.clone()))) + } +} + +pub fn import_module_start(mut statement_text: String, mut at: i64, mut limit: i64) -> i64 { + loop { + if (at.clone() >= limit.clone()) { + break at.clone(); + } else { + if is_import_whitespace(v1_rt::char_at(&statement_text, at.clone())) { + { + let __tco_0 = (at + 1); + at = __tco_0; + continue; + } + } else { + break at.clone(); + } + } + } +} + +pub fn statement_text_names_module(statement_text: String, imported_module: String) -> bool { + { + let text_length = v1_rt::string_length(&statement_text); + let module_start = import_module_start( + statement_text.clone(), + v1_rt::string_length(&import_statement_keyword()), + text_length.clone(), + ); + let module_end = (module_start.clone() + v1_rt::string_length(&imported_module)); + ((((v1_rt::string_length(&imported_module) > 0) + && (module_end.clone() <= text_length.clone())) + && (v1_rt::substring(&statement_text, module_start.clone(), module_end.clone()) + == imported_module.clone())) + && ((module_end.clone() == text_length.clone()) + || is_import_module_boundary(v1_rt::char_at(&statement_text, module_end.clone())))) + } +} + +pub fn import_strip_step( + progress: Rc, + source_file: String, + source: String, + source_length: i64, + statement: Rc, +) -> Rc { + match (*progress.clone()).clone() { + ImportStripProgress::ImportStripHalted { + refusal: refusal, .. + } => Rc::new(ImportStripProgress::ImportStripHalted { + refusal: refusal.clone(), + }), + ImportStripProgress::ImportStripScanning { + cursor, + kept_segments, + removed_statements, + .. + } => { + if !(statement.span.clone().file.clone() == source_file.clone()) { + Rc::new(ImportStripProgress::ImportStripHalted { + refusal: Rc::new(ImportStripRefusal::ImportSpanNamesAnotherFile { + span_file: statement.span.clone().file.clone(), + source_file: source_file.clone(), + }), + }) + } else { + if ((statement.span.clone().start.clone() < 0) + || (statement.span.clone().end.clone() > source_length.clone())) + { + Rc::new(ImportStripProgress::ImportStripHalted { + refusal: Rc::new(ImportStripRefusal::ImportSpanOutsideSource { + span: statement.span.clone(), + source_length: source_length.clone(), + }), + }) + } else { + if (statement.span.clone().end.clone() <= statement.span.clone().start.clone()) + { + Rc::new(ImportStripProgress::ImportStripHalted { + refusal: Rc::new(ImportStripRefusal::ImportSpanIsEmpty { + span: statement.span.clone(), + }), + }) + } else { + if (statement.span.clone().start.clone() < cursor.clone()) { + Rc::new(ImportStripProgress::ImportStripHalted { + refusal: Rc::new( + ImportStripRefusal::ImportSpansOutOfAscendingOrder { + previous_end: cursor.clone(), + next_start: statement.span.clone().start.clone(), + }, + ), + }) + } else { + { + let statement_text = v1_rt::substring( + &source, + statement.span.clone().start.clone(), + statement.span.clone().end.clone(), + ); + if !statement_text_opens_with_the_import_keyword( + statement_text.clone(), + ) { + Rc::new(ImportStripProgress::ImportStripHalted { + refusal: Rc::new(ImportStripRefusal::ImportSpanDoesNotDelimitAnImportStatement { + span: statement.span.clone(), + source_text: statement_text.clone(), +}), +}) + } else { + if !statement_text_names_module( + statement_text.clone(), + statement.imported_module.clone(), + ) { + Rc::new(ImportStripProgress::ImportStripHalted { + refusal: Rc::new(ImportStripRefusal::ImportSpanDoesNotNameItsModule { + span: statement.span.clone(), + imported_module: statement.imported_module.clone(), + source_text: statement_text.clone(), +}), +}) + } else { + Rc::new(ImportStripProgress::ImportStripScanning { + cursor: statement.span.clone().end.clone(), + kept_segments: v1_rt::rc_list_push( + kept_segments.clone(), + v1_rt::substring( + &source, + cursor.clone(), + statement.span.clone().start.clone(), + ), + ), + removed_statements: v1_rt::rc_list_push( + removed_statements.clone(), + statement_text.clone(), + ), + }) + } + } + } + } + } + } + } + } + } +} + +pub fn strip_import_statements( + source_file: String, + source: String, + statements: Rc>>, +) -> Rc { + { + let source_length = v1_rt::string_length(&source); + let scanned = statements.iter().cloned().fold( + Rc::new(ImportStripProgress::ImportStripScanning { + cursor: 0, + kept_segments: Rc::new(vec![]), + removed_statements: Rc::new(vec![]), + }), + |progress: Rc, statement: Rc| { + import_strip_step( + progress, + source_file.clone(), + source.clone(), + source_length.clone(), + statement.clone(), + ) + }, + ); + match (*scanned.clone()).clone() { + ImportStripProgress::ImportStripHalted { + refusal: refusal, .. + } => Rc::new(ImportStripOutcome::ImportStripRefused { + refusal: refusal.clone(), + }), + ImportStripProgress::ImportStripScanning { + cursor, + kept_segments, + removed_statements, + .. + } => Rc::new(ImportStripOutcome::ImportStatementsStripped { + rewritten_source: v1_rt::rc_list_push( + kept_segments.clone(), + v1_rt::substring(&source, cursor.clone(), source_length.clone()), + ) + .join(&"".to_string()), + removed_statements: removed_statements.clone(), + }), + } + } +} + +pub fn strip_one_import_statement( + source_file: String, + source: String, + statement: Rc, +) -> Rc { + strip_import_statements( + source_file.clone(), + source.clone(), + Rc::new(vec![statement.clone()]), + ) +} diff --git a/src/v1/stage0/src/v1_compiler_infer_method.rs b/src/v1/stage0/src/v1_compiler_infer_method.rs index 8130370860e..21b9ff8235b 100644 --- a/src/v1/stage0/src/v1_compiler_infer_method.rs +++ b/src/v1/stage0/src/v1_compiler_infer_method.rs @@ -1071,6 +1071,20 @@ pub fn builtin_function_registry() -> Rc>> }), })]), returns: type_variable_node("multi_module_compile_fixture_result".to_string()), + })); + __m.insert("parsed_import_statements".to_string(), Rc::new(BuiltinSignature { + params: Rc::new(vec![Rc::new(BuiltinParam { + name: "file".to_string(), + ty: Rc::new(AlgebraTypeTemplate::NamedTemplate { + name: "String".to_string(), + }), + }), Rc::new(BuiltinParam { + name: "source".to_string(), + ty: Rc::new(AlgebraTypeTemplate::NamedTemplate { + name: "String".to_string(), + }), + })]), + returns: type_variable_node("parsed_import_statements_result".to_string()), })); __m.insert("observe_declared_import_closure_symbol_binding".to_string(), Rc::new(BuiltinSignature { params: Rc::new(vec![Rc::new(BuiltinParam { diff --git a/src/v1/stage0/src/v1_compiler_parse.rs b/src/v1/stage0/src/v1_compiler_parse.rs index ff9d9e584c7..f59032416ec 100644 --- a/src/v1/stage0/src/v1_compiler_parse.rs +++ b/src/v1/stage0/src/v1_compiler_parse.rs @@ -10,6 +10,10 @@ use self::ParserHelperIdentity::*; use self::ParserResultWitness::*; pub use crate::extdeps_languages_dag_syntax::{dag_non_name_keywords, dag_syntax_spec}; pub use crate::std_algebra::FreeMonoid; +use crate::std_import::ParsedImportStatements::{ + ImportStatementParseRefused, ImportStatementsParsed, +}; +pub use crate::std_import::{ParsedImportStatement, ParsedImportStatements}; use crate::std_occurrence_identity::NodeOccurrenceIdentity::{ OccurrenceMinted, OccurrenceProjected, OccurrenceSynthetic, }; @@ -39,7 +43,7 @@ use crate::std_syntax::LiteralValue::{LitBool, LitFloat, LitInt, LitNull, LitSym pub use crate::std_syntax::{ BinOp, BodyKind, ItemForm, ItemFormKind, LiteralValue, OperatorSpec, SyntaxSpec, }; -pub use crate::std_types::SourceSpan; +pub use crate::std_types::{NonEmptyStr, SourceSpan}; use crate::v1_rt; use crate::v1_rt::{VecCompat, VecJoin}; pub use crate::v1_std_core::make_file_span; @@ -3330,6 +3334,29 @@ pub fn occurrence_transport_from_parse_context(ctx: Rc) -> Rc>>, + source_indices: Rc>>, + intern_table: Rc, + occurrence_base: Rc, + heads_only: bool, +) -> Rc { + Rc::new(ParseContext { + source_indices: source_indices.clone(), + intern_table: crate::v1_std_core::pre_intern_tokens(tokens.clone(), intern_table.clone()), + occurrence_allocator: crate::std_occurrence_identity::occurrence_id_allocator_advance_to( + occurrence_base.allocator.clone(), + intern_table.authored_token_ordinals.clone(), + ), + occurrence_index: Some(Rc::new(OccurrenceIndex { + entries: Rc::new(vec![]), + })), + declaration_occurrences: Some(Rc::new(vec![])), + reference_occurrences: Some(Rc::new(vec![])), + heads_only: heads_only.clone(), + }) +} + pub fn parse_with_table_at( tokens: Rc>>, source_indices: Rc>>, @@ -3343,19 +3370,13 @@ pub fn parse_with_table_at( occurrence_base.allocator.clone(), intern_table.authored_token_ordinals.clone(), ); - let pre_interned = - crate::v1_std_core::pre_intern_tokens(tokens.clone(), intern_table.clone()); - let ctx = Rc::new(ParseContext { - source_indices: source_indices.clone(), - intern_table: pre_interned.clone(), - occurrence_allocator: occurrence_allocator.clone(), - occurrence_index: Some(Rc::new(OccurrenceIndex { - entries: Rc::new(vec![]), - })), - declaration_occurrences: Some(Rc::new(vec![])), - reference_occurrences: Some(Rc::new(vec![])), - heads_only: heads_only.clone(), - }); + let ctx = parse_context_for_tokens( + tokens.clone(), + source_indices.clone(), + intern_table.clone(), + occurrence_base.clone(), + heads_only.clone(), + ); let r = parse_module(token_stream_new(tokens.clone()), ctx.clone()); if has_err(r.err.clone()) { { @@ -3640,6 +3661,147 @@ pub fn parse_imports_acc( } } +pub fn last_consumed_token_end( + mut all: Rc>>, + mut from: i64, + mut until: i64, + mut end: Option, +) -> Option { + loop { + if (from.clone() >= until.clone()) { + break end; + } else { + match all.clone().get((from.clone()) as usize).cloned() { + std::option::Option::None => { + let __tco_0 = (from + 1); + from = __tco_0; + continue; + } + Some(t) => { + if is_newline_shape(t.shape.clone()) { + { + let __tco_0 = (from + 1); + from = __tco_0; + continue; + } + } else { + { + let __tco_0 = (from + 1); + let __tco_1 = Some(t.span.clone().end.clone()); + from = __tco_0; + end = __tco_1; + continue; + } + } + } + } + } + } +} + +pub fn parse_import_statement_extents_acc( + mut tokens: Rc, + mut ctx: Rc, + mut acc: Rc>>, +) -> Rc { + loop { + tokens = skip_newlines(tokens.clone()); + if tok_is_keyword(token_stream_first(tokens.clone()), "import".to_string()) { + let before = tokens.pos.clone(); + let r = parse_import(tokens.clone(), ctx.clone()); + if has_err(r.err.clone()) { + return Rc::new(ParsedImportStatements::ImportStatementParseRefused { + cause: "an import statement did not parse".to_string(), + }); + } + match token_stream_first(tokens.clone()) { + std::option::Option::None => { + break Rc::new(ParsedImportStatements::ImportStatementParseRefused { + cause: "an import statement parsed from no token".to_string(), + }); + } + Some(first) => { + match last_consumed_token_end( + tokens.all.clone(), + before.clone(), + r.tokens.clone().pos.clone(), + std::option::Option::None, + ) { + std::option::Option::None => { + break Rc::new(ParsedImportStatements::ImportStatementParseRefused { + cause: "an import statement parsed without consuming a token" + .to_string(), + }); + } + Some(end) => { + let __tco_0 = r.tokens.clone(); + let __tco_1 = parse_context_after_node(r.ctx.clone(), r.import.clone()); + let __tco_2 = v1_rt::rc_list_push( + acc, + Rc::new(ParsedImportStatement { + span: Rc::new(SourceSpan { + file: first.span.clone().file.clone(), + start: first.span.clone().start.clone(), + end: end.clone(), + }), + imported_module: r.import.clone().name.clone(), + }), + ); + tokens = __tco_0; + ctx = __tco_1; + acc = __tco_2; + continue; + } + } + } + } + } else { + break Rc::new(ParsedImportStatements::ImportStatementsParsed { + statements: acc.clone(), + }); + } + } +} + +pub fn parse_import_statement_extents( + tokens: Rc>>, + source_indices: Rc>>, + intern_table: Rc, +) -> Rc { + { + let ctx = parse_context_for_tokens( + tokens.clone(), + source_indices.clone(), + intern_table.clone(), + intern_table.authored_token_ordinals.clone(), + false, + ); + let stream = skip_newlines(token_stream_new(tokens.clone())); + let r = expect( + stream.clone(), + Rc::new(ExpectedToken::ExpectKeyword { + text: "module".to_string(), + }), + ); + if has_err(r.err.clone()) { + return Rc::new(ParsedImportStatements::ImportStatementParseRefused { + cause: "source does not open with a module declaration".to_string(), + }); + } + let r = parse_dotted_ident(r.tokens.clone()); + if has_err(r.err.clone()) { + return Rc::new(ParsedImportStatements::ImportStatementParseRefused { + cause: "module declaration does not name a dotted module path".to_string(), + }); + } + parse_import_statement_extents_acc( + skip_newlines(r.tokens.clone()), + ctx.clone(), + Rc::new(vec![]), + ) + } +} + pub fn parse_items(tokens: Rc, ctx: Rc) -> Rc { parse_items_acc(tokens.clone(), ctx.clone(), Rc::new(vec![])) } diff --git a/src/v1/stage0/src/v1_gunbc_parsed_import_statements.rs b/src/v1/stage0/src/v1_gunbc_parsed_import_statements.rs new file mode 100644 index 00000000000..f805e7ab38e --- /dev/null +++ b/src/v1/stage0/src/v1_gunbc_parsed_import_statements.rs @@ -0,0 +1,27 @@ +// Generated by v1 compiler -- do not edit. +// Source module: v1.gunbc.parsed_import_statements + +pub use crate::std_import::ParsedImportStatements; +use crate::std_import::ParsedImportStatements::*; +pub use crate::v1_compiler_parse::parse_import_statement_extents; +pub use crate::v1_compiler_tokenize::tokenize; +use crate::v1_rt; +use crate::v1_rt::{VecCompat, VecJoin}; +pub use crate::v1_std_core::NewlineIndex; +pub use crate::v1_std_core::{build_newline_index, empty_intern_table}; +use crate::NonEmptyBTreeSet; +use crate::NonEmptyVec; +use im::{vector as vec, HashMap, OrdSet as BTreeSet, Vector as Vec}; +use std::rc::Rc; + +pub fn parsed_import_statements(file: String, source: String) -> Rc { + crate::v1_compiler_parse::parse_import_statement_extents( + crate::v1_compiler_tokenize::tokenize(source.clone(), file.clone()), + v1_rt::rc_map_insert( + v1_rt::rc_empty_map::>(), + file.clone(), + crate::v1_std_core::build_newline_index(file.clone(), source.clone()), + ), + crate::v1_std_core::empty_intern_table(), + ) +} diff --git a/src/v1/stage0/src/v1_interpreter.rs b/src/v1/stage0/src/v1_interpreter.rs index 6d967c8c06a..697b9f739c9 100644 --- a/src/v1/stage0/src/v1_interpreter.rs +++ b/src/v1/stage0/src/v1_interpreter.rs @@ -11490,6 +11490,62 @@ fn subject_module_value( } } +/// Encode one source's parsed import statements: the parser's delimited spans, or the typed +/// refusal. A source that did not parse and a source with no imports are different values here +/// because they are different facts, and a caller that stripped nothing from the first would +/// report a clean rewrite of a file it never read. +fn parsed_import_statements_value( + outcome: &crate::std_import::ParsedImportStatements, + ctx: &InterpContext, +) -> Value { + use crate::std_import::ParsedImportStatements as P; + match outcome { + P::ImportStatementsParsed { statements } => Value::Variant { + type_name: ctx.sym("ParsedImportStatements"), + variant_name: ctx.sym("ImportStatementsParsed"), + fields: Rc::new(sorted_fields(vec![( + ctx.sym("statements"), + list_value( + statements + .iter() + .map(|statement| Value::Record { + type_name: ctx.sym("ParsedImportStatement"), + fields: Rc::new(sorted_fields(vec![ + ( + ctx.sym("span"), + Value::Record { + type_name: ctx.sym("SourceSpan"), + fields: Rc::new(sorted_fields(vec![ + ( + ctx.sym("file"), + str_value(statement.span.file.clone()), + ), + (ctx.sym("start"), Value::Int(statement.span.start)), + (ctx.sym("end"), Value::Int(statement.span.end)), + ])), + }, + ), + ( + ctx.sym("imported_module"), + str_value(statement.imported_module.clone()), + ), + ])), + }) + .collect::>(), + ), + )])), + }, + P::ImportStatementParseRefused { cause } => Value::Variant { + type_name: ctx.sym("ParsedImportStatements"), + variant_name: ctx.sym("ImportStatementParseRefused"), + fields: Rc::new(sorted_fields(vec![( + ctx.sym("cause"), + str_value(cause.clone()), + )])), + }, + } +} + fn namespace_structural_observation_admissions_value( compiled_module: &crate::std_reference_binding_observation::StructuralObservationSubjectModule, admissions: &[Rc], @@ -17156,6 +17212,16 @@ macro_rules! v1_builtin_arms { ))), arm "free_call.doc_graph_doc_count" { "doc_graph_doc_count" } => Ok(Some(Value::Int(crate::cli_run::doc_graph_doc_count()))), + arm "free_call.parsed_import_statements" { "parsed_import_statements" } => { + let file = expect_str($positional.first().copied(), $name)?; + let source = expect_str($positional.get(1).copied(), $name)?; + let observed = + crate::v1_gunbc_parsed_import_statements::parsed_import_statements( + file, source, + ); + Ok(Some(parsed_import_statements_value(&observed, $ctx))) + }, + arm "free_call.namespace_structural_observation_admissions" { "namespace_structural_observation_admissions" } => { let file = expect_str($positional.first().copied(), $name)?; let source = expect_str($positional.get(1).copied(), $name)?; diff --git a/src/v1/stage0/src/v1_interpreter_dispatch_generated.rs b/src/v1/stage0/src/v1_interpreter_dispatch_generated.rs index ace5511e840..42c55cd0a57 100644 --- a/src/v1/stage0/src/v1_interpreter_dispatch_generated.rs +++ b/src/v1/stage0/src/v1_interpreter_dispatch_generated.rs @@ -92,6 +92,7 @@ pub enum EvalBuiltinArm { FreeCallDocGraphAdmittedRootCount, FreeCallDocGraphDanglingLinkCount, FreeCallDocGraphDocCount, + FreeCallParsedImportStatements, FreeCallNamespaceStructuralObservationAdmissions, FreeCallCompileDagRustEmitCheck, FreeCallCompileDagDiagnosticCensus, @@ -225,6 +226,7 @@ pub fn lookup_eval_builtin_inner(spelling: &str) -> Option { "doc_graph_admitted_root_count" => Some(EvalBuiltinArm::FreeCallDocGraphAdmittedRootCount), "doc_graph_dangling_link_count" => Some(EvalBuiltinArm::FreeCallDocGraphDanglingLinkCount), "doc_graph_doc_count" => Some(EvalBuiltinArm::FreeCallDocGraphDocCount), + "parsed_import_statements" => Some(EvalBuiltinArm::FreeCallParsedImportStatements), "namespace_structural_observation_admissions" => Some(EvalBuiltinArm::FreeCallNamespaceStructuralObservationAdmissions), "compile_dag_rust_emit_check" => Some(EvalBuiltinArm::FreeCallCompileDagRustEmitCheck), "compile_dag_diagnostic_census" => Some(EvalBuiltinArm::FreeCallCompileDagDiagnosticCensus), @@ -356,6 +358,7 @@ macro_rules! eval_builtin_inner_arm { ("free_call.doc_graph_admitted_root_count") => { $crate::v1_interpreter_dispatch_generated::EvalBuiltinArm::FreeCallDocGraphAdmittedRootCount }; ("free_call.doc_graph_dangling_link_count") => { $crate::v1_interpreter_dispatch_generated::EvalBuiltinArm::FreeCallDocGraphDanglingLinkCount }; ("free_call.doc_graph_doc_count") => { $crate::v1_interpreter_dispatch_generated::EvalBuiltinArm::FreeCallDocGraphDocCount }; + ("free_call.parsed_import_statements") => { $crate::v1_interpreter_dispatch_generated::EvalBuiltinArm::FreeCallParsedImportStatements }; ("free_call.namespace_structural_observation_admissions") => { $crate::v1_interpreter_dispatch_generated::EvalBuiltinArm::FreeCallNamespaceStructuralObservationAdmissions }; ("free_call.compile_dag_rust_emit_check") => { $crate::v1_interpreter_dispatch_generated::EvalBuiltinArm::FreeCallCompileDagRustEmitCheck }; ("free_call.compile_dag_diagnostic_census") => { $crate::v1_interpreter_dispatch_generated::EvalBuiltinArm::FreeCallCompileDagDiagnosticCensus };