Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
136 changes: 110 additions & 26 deletions dag/gunbc/self_host_compile_phase_frontier.dag
Original file line number Diff line number Diff line change
Expand Up @@ -139,11 +139,25 @@ fn diagnostic_is_phase(d: RustcCodedDiagnostic, phase: RustcPhase) -> Bool {
}
}

// THE MEMBERSHIP TEST IS AGAINST THE PREVIOUS ELEMENT, NOT THE WHOLE ACCUMULATOR, and the sort is
// what makes that sound. Scanning the accumulated set per element is a quadratic fold over a
// population that only grows -- 233 identities on the genesis receipt today -- and DESIGN section 6
// fixes a proven cost-shape defect regardless of the realized n, because "n is small here" stops
// being true the moment a receipt is added. Sorting first puts equal identities adjacent, so one
// comparison against the last kept element decides membership.
//
// BOTH CALLERS ARE ORDER-INDIFFERENT, WHICH IS WHY THE SORT IS FREE RATHER THAN A SEMANTIC CHANGE:
// `canonical_identity_set` sorts its result anyway, and `census_population_is_duplicate_free` reads
// only the count. Encounter order was never observable through either, and the row above states why
// it must not be: a board is a census at identity grain, not a transcript of emission order.
fn deduplicate_identities(identities: List<String>) -> List<String> {
fold(
identities,
identities |> sort_by(identity => identity),
init: [],
f: (unique, identity) => if any(unique, prior => prior == identity) { unique } else { concat(unique, [identity]) }
f: (unique, identity) => match unique |> last {
Present { value: prior } => if prior == identity { unique } else { concat(unique, [identity]) }
Absent => [identity]
}
)
}

Expand All @@ -152,8 +166,14 @@ fn deduplicate_identities(identities: List<String>) -> List<String> {
// meaning. Give that set one representation here so structural board equality cannot accidentally
// make cargo/rustfmt emission order semantic. Duplicate refusal remains separate and upstream:
// canonicalizing a board does not canonicalize either raw population before coherence checks it.
// THE SORT LIVES IN `deduplicate_identities` AND IS NOT REPEATED HERE. Dedup needs the population
// ordered to decide membership against one neighbour instead of the whole accumulator, so its
// result is ALREADY canonical; sorting it again was a second sort of the same list for a property
// the first one established. On a 233-identity receipt that duplicate sort was the single largest
// cost in the fold, which is what makes it a section 2 defect rather than a tidy-up: the second
// demand for an ordering was satisfiable from the first.
fn canonical_identity_set(identities: List<String>) -> List<String> {
deduplicate_identities(identities: identities) |> sort_by(identity => identity)
deduplicate_identities(identities: identities)
}

// A census population carries no repeats, and this predicate is where that is decided. Refusal is
Expand Down Expand Up @@ -333,32 +353,95 @@ fn census_identity_code(identity: String) -> String? {
rustc_diagnostic_row_code(row: identity)
}

fn census_identity_phase(identity: String) -> RustcErrorCodePhase {
match census_identity_code(identity: identity) {
fn census_phase_of_code(code: String?) -> RustcErrorCodePhase {
match code {
Absent => ErrorCodeUnplaced { code: "malformed-census-identity", index: extdeps.languages.rust.compiler_phases.rustc_error_index_authority }
Present { value: code } => rustc_error_code_phase(code: code)
Present { value: c } => rustc_error_code_phase(code: c)
}
}

fn census_identities_for_phase(identities: List<String>, phase: RustcPhase) -> List<String> {
canonical_identity_set(identities: identities |> filter(identity => match census_identity_phase(identity: identity) {
ErrorCodePlaced { phase: placed, authority: _ } => rustc_phase_order(phase: placed) == rustc_phase_order(phase: phase)
ErrorCodeUnplaced { code: _, index: _ } => false
}))
fn census_identity_phase(identity: String) -> RustcErrorCodePhase {
census_phase_of_code(code: census_identity_code(identity: identity))
}

// The carried classification of one identity. It exists so the five buckets below are DERIVED from
// one pass rather than re-deciding the same question per bucket; the rationale for that lives on
// `census_identity_partition`, which is the function it governs.
type ClassifiedIdentity {
identity: String
phase: RustcErrorCodePhase
codeless: Bool
}

type CensusIdentityPartition {
resolve: List<String>
typeck: List<String>
borrowck: List<String>
unplaced: List<String>
codeless: List<String>
}

fn classify_identity(identity: String) -> ClassifiedIdentity {
let code = census_identity_code(identity: identity)
ClassifiedIdentity {
identity: identity,
phase: census_phase_of_code(code: code),
codeless: match code {
Present { value: c } => c == "uncoded"
Absent => false
}
}
}

fn census_unplaced_identities(identities: List<String>) -> List<String> {
canonical_identity_set(identities: identities |> filter(identity => match census_identity_phase(identity: identity) {
ErrorCodePlaced { phase: _, authority: _ } => false
ErrorCodeUnplaced { code: _, index: _ } => true
}))
fn classified_at_phase(classified: List<ClassifiedIdentity>, phase: RustcPhase) -> List<String> {
classified
|> filter(c => match c.phase {
ErrorCodePlaced { phase: p, authority: _ } => rustc_phase_order(phase: p) == rustc_phase_order(phase: phase)
ErrorCodeUnplaced { code: _, index: _ } => false
})
|> map(c => c.identity)
}

fn census_codeless_identities(identities: List<String>) -> List<String> {
canonical_identity_set(identities: identities |> filter(identity => match census_identity_code(identity: identity) {
Present { value: code } => code == "uncoded"
Absent => false
}))
// THE CLASSIFICATION IS CARRIED, NOT RECOMPUTED, AND THE BUCKETS ARE BUILT BY `filter` RATHER THAN
// BY APPENDING IN A FOLD. Each bucket below used to be its own `filter` over the whole identity
// list, so a 233-identity receipt was traversed and re-classified six times to answer six questions
// about the SAME classification -- DESIGN section 2's authored duplication rather than a caching
// obligation, since the six demands share a visible ancestor. Classifying once and deriving the
// buckets removes the recomputation.
//
// THE FIRST VERSION OF THIS FUNCTION FOLDED WITH `concat(acc.bucket, [identity])` PER ELEMENT, and
// that is a copied accumulator across five buckets -- the same section 6 defect this change exists
// to remove, reintroduced one line below the fix. Found in review on gunbc#10038. Building each
// bucket with `filter` over the carried classification keeps one pass per bucket over an already
// classified list and no quadratic append.
//
// NO MEASUREMENT IS QUOTED IN THIS ANNOTATION, AND THE REASON IS SHARPER THAN SECTION 6's
// CITE-THE-INSTRUMENT RULE. An earlier draft carried millisecond figures taken from probes that
// DELETED the work being timed; the witnesses failed on those runs, so the numbers measured the
// cost of failing early rather than the cost of the fold. A transcribed number cannot carry the
// pass/fail context that decides whether it means anything, which is why the rule names the
// producer instead: `claim_batch` over this module's witness entry re-derives the per-witness cost,
// and its own PASS/FAIL count is what qualifies the reading. The section 6 argument above needs no
// number at all -- a copied accumulator and a repeated pass are fixed regardless of realized n.
//
// THE BUCKETS ARE NOT DISJOINT AND THE DERIVATION MUST NOT ASSUME THEY ARE. `codeless` is decided on
// the code spelling and `unplaced` on whether the code resolves in the error index, so an identity
// carrying `uncoded` belongs to BOTH. A single match with one arm per identity would have silently
// dropped one of the two memberships.
fn census_identity_partition(identities: List<String>) -> CensusIdentityPartition {
let classified = identities |> map(identity => classify_identity(identity: identity))
CensusIdentityPartition {
resolve: classified_at_phase(classified: classified, phase: Resolve),
typeck: classified_at_phase(classified: classified, phase: Typeck),
borrowck: classified_at_phase(classified: classified, phase: Borrowck),
unplaced: classified
|> filter(c => match c.phase {
ErrorCodePlaced { phase: _, authority: _ } => false
ErrorCodeUnplaced { code: _, index: _ } => true
})
|> map(c => c.identity),
codeless: classified |> filter(c => c.codeless) |> map(c => c.identity)
}
}

// Persisted receipts are folded from the raw identity rows emitted by the same instrument. This
Expand All @@ -367,11 +450,12 @@ fn phase_board_from_census_identities(
parse_observation: RustfmtParseObservation,
raw_identities: List<String>
) -> PersistedPhaseBoardFold {
let resolve_ids = census_identities_for_phase(identities: raw_identities, phase: Resolve)
let typeck_ids = census_identities_for_phase(identities: raw_identities, phase: Typeck)
let borrowck_ids = census_identities_for_phase(identities: raw_identities, phase: Borrowck)
let unplaced = census_unplaced_identities(identities: raw_identities)
let codeless = census_codeless_identities(identities: raw_identities)
let partition = census_identity_partition(identities: raw_identities)
let resolve_ids = canonical_identity_set(identities: partition.resolve)
let typeck_ids = canonical_identity_set(identities: partition.typeck)
let borrowck_ids = canonical_identity_set(identities: partition.borrowck)
let unplaced = canonical_identity_set(identities: partition.unplaced)
let codeless = canonical_identity_set(identities: partition.codeless)
let unique_errors = canonical_identity_set(identities: raw_identities)
let parse_identities = canonical_identity_set(identities: parse_observation.refused_files)
let parse_errors = count(parse_identities)
Expand Down
Loading