Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
219 changes: 205 additions & 14 deletions dag/gunbc/floor_cost_distribution.dag
Original file line number Diff line number Diff line change
Expand Up @@ -23,10 +23,17 @@ import std.measure { Millisecond, millisecond, millisecond_count }
// THE INPUT IS THE `required-floor-claim-cost` ARTIFACT that every required-floor run uploads --
// the COMPLETE executed population, not the 25-row `[over-cost]` preview the job log prints.

// `eval_steps` IS THE THIRD COLUMN THIS ANALYSIS NEEDS, and the only one of the three that is not
// a clock. Evaluator steps are counted by the interpreter and netted by the same shared-fill rule
// the CPU figure is, so a row's step count is a property of what it evaluated and carries no term
// for the machine it evaluated on. It is what lets `identity_inflation_permille_at_equal_work`
// below separate "this row got more expensive" from "this row got inflated" without choosing a
// control population at all.
type FloorCostRow {
identity: String
module: String
cpu_ms: Millisecond
eval_steps: Int
}

// One run's population, named by its run id so a finding can cite which runs it rests on.
Expand All @@ -35,40 +42,114 @@ type RunCost {
rows: List<FloorCostRow>
}

// COLUMNS ARE LOCATED BY NAME, NOT BY POSITION, AND THIS IS NOT STYLE.
//
// DESIGN section 3's standing rule -- cite the symbol, never the position -- applies to a TSV
// exactly as it applies to a source file, and for the same reason: any insertion ABOVE the cited
// index silently invalidates it. That is not hypothetical here. The `eval_steps` column was
// INSERTED into this artifact between vintages: run 33661252708's header is
// `identity module outcome verdict_reached wall_ms cpu_ms cost_line_ms` and run 33699540412's is
// the same with `eval_steps` before `cost_line_ms`. A parser keyed on index 6 reads the older
// artifact's `cost_line_ms` -- the constant 100 -- as a step count, which is not an error, it is a
// PLAUSIBLE NUMBER: the resulting microseconds-per-step figures came out near 2500 instead of near
// 2, off by three orders of magnitude and still perfectly well-formed. Measured, not imagined:
// three of eleven sampled runs produced exactly that before the header was consulted.
//
// So the header IS the schema and a missing column REFUSES. `Absent` here means "this artifact
// does not carry the column this analysis needs", which is a fact about the artifact and must
// reach the caller as one, rather than being defaulted into a zero or a constant that flows into
// every downstream percentile.
type ClaimCostColumns {
identity: Int
module: Int
cpu_ms: Int
eval_steps: Int
}

fn claim_cost_column_index(header_fields: List<String>, name: String) -> Int? {
header_fields
|> enumerate
|> filter(e => e.second == name)
|> map(e => e.first)
|> first
}

// THE HEADER LINE IS THE ONE THAT NAMES `identity` FIRST. The artifact's `#`-prefixed summary
// precedes it, and a data row can never be mistaken for it because no claim identity is the
// literal string "identity".
fn claim_cost_columns(header_line: String) -> ClaimCostColumns? {
let fields = split(s: header_line, delimiter: "\t")
match claim_cost_column_index(header_fields: fields, name: "identity") {
Absent => none
Present { value: id } => match claim_cost_column_index(header_fields: fields, name: "module") {
Absent => none
Present { value: md } => match claim_cost_column_index(header_fields: fields, name: "cpu_ms") {
Absent => none
Present { value: cpu } => match claim_cost_column_index(header_fields: fields, name: "eval_steps") {
Absent => none
Present { value: steps } => Present {
value: ClaimCostColumns { identity: id, module: md, cpu_ms: cpu, eval_steps: steps }
}
}
}
}
}
}

// A TSV LINE IS EITHER A ROW OR IT IS NOT, AND "NOT" IS NEVER A FABRICATED ROW. The artifact
// carries a `#`-prefixed summary line and a header line before the data, and a short or
// unparseable line is refused rather than defaulted to zero -- a zero would sit silently at the
// bottom of every band and inflate the safe population.
fn parse_claim_cost_line(line: String) -> FloorCostRow? {
fn parse_claim_cost_line(line: String, columns: ClaimCostColumns) -> FloorCostRow? {
if starts_with(s: line, prefix: "#") {
none
} else {
let fields = split(s: line, delimiter: "\t")
if count(fields) < 6 {
let widest = max_int(a: max_int(a: columns.identity, b: columns.module), b: max_int(a: columns.cpu_ms, b: columns.eval_steps))
if count(fields) <= widest {
none
} else if fields[0] == "identity" {
} else if fields[columns.identity] == "identity" {
none
} else {
match parse_int(s: fields[5]) {
match parse_int(s: fields[columns.cpu_ms]) {
Absent => none
Present { value: cpu } => Present {
value: FloorCostRow {
identity: fields[0],
module: fields[1],
cpu_ms: millisecond(count: cpu)
Present { value: cpu } => match parse_int(s: fields[columns.eval_steps]) {
Absent => none
Present { value: steps } => Present {
value: FloorCostRow {
identity: fields[columns.identity],
module: fields[columns.module],
cpu_ms: millisecond(count: cpu),
eval_steps: steps
}
}
}
}
}
}
}

fn max_int(a: Int, b: Int) -> Int {
if a > b { a } else { b }
}

// AN ARTIFACT WHOSE HEADER LACKS A NEEDED COLUMN YIELDS NO ROWS, WHICH IS THE WHOLE POPULATION
// REFUSING RATHER THAN A SILENTLY THINNER ONE. A caller that wants to distinguish "old vintage"
// from "empty run" reads `claim_cost_columns` itself; this function answers only the question it
// is asked, and answers it with nothing rather than with rows built on a guessed layout.
fn parse_claim_cost_tsv(text: String) -> List<FloorCostRow> {
split(s: text, delimiter: "\n")
|> flat_map(line => match parse_claim_cost_line(line: line) {
Absent => []
Present { value: row } => [row]
})
let lines = split(s: text, delimiter: "\n")
match lines |> filter(l => starts_with(s: l, prefix: "identity\t")) |> first {
Absent => []
Present { value: header } => match claim_cost_columns(header_line: header) {
Absent => []
Present { value: columns } =>
lines |> flat_map(line => match parse_claim_cost_line(line: line, columns: columns) {
Absent => []
Present { value: row } => [row]
})
}
}
}

// THE BAND HISTOGRAM (question 1: is the population near the ceiling dense).
Expand Down Expand Up @@ -224,3 +305,113 @@ fn implied_clean_run_budget(ceiling_ms: Millisecond, inflation_permille: Int) ->
millisecond(count: (millisecond_count(m: ceiling_ms) * 1000) / inflation_permille)
}
}

// INFLATION WITH THE WORK CONFOUND REMOVED BY CONSTRUCTION, WHICH IS WHY IT NEEDS NO CONTROL.
//
// `identity_inflation_permille` above pairs a row against itself across two runs, which is already
// the right pairing -- but a row whose cost moved because the TREE changed is inflation and cost
// change mixed, and separating them normally means choosing a control population and defending
// the choice. Every such choice made in gunbc#10094 had to be defended and two were withdrawn.
//
// Restricting the pairing to rows whose `eval_steps` are EQUAL removes the confound instead of
// modelling it: the evaluator executed the same number of steps in both runs, so the only thing
// that could have changed did not, and any remaining cpu movement is inflation. Reach is not
// assumed -- `count` the result against `identity_inflation_permille`'s to see what fraction of
// the population the discriminator covers, because a clean instrument over three rows is not a
// measurement of a corpus.
//
// TWO ROWS CAN SHARE A STEP COUNT WITHOUT SHARING WORK, and the honest statement of what this
// establishes accounts for it: equal steps means equal EVALUATOR work, not equal host work. A row
// that spends its time inside one opaque host call reports few steps in both runs and is admitted
// here while its real variation lives outside the evaluator. `v2.workflow.required_floor` already
// carries a declared rung drop for exactly that region, so the population this instrument is
// cleanest over is the one that is CPU-bound in the evaluator.
fn identity_inflation_permille_at_equal_work(
base: RunCost,
other: RunCost,
min_base_ms: Millisecond
) -> List<Int> {
base.rows
|> filter(b => millisecond_count(m: b.cpu_ms) >= millisecond_count(m: min_base_ms))
|> flat_map(b => match other.rows |> filter(o => o.identity == b.identity && o.eval_steps == b.eval_steps) |> first {
Absent => []
Present { value: o } => [(millisecond_count(m: o.cpu_ms) * 1000) / millisecond_count(m: b.cpu_ms)]
})
}

// HOW MANY RUNS ONE IDENTITY CROSSED IN, which is the question that separates a property of the
// ROW from a property of the RUN.
type CrossingRecurrence {
identity: String
runs_crossed: Int
}

// THE RECURRENCE JOIN. A ceiling that refuses the whole run makes "which rows crossed" a per-run
// set, and reading one such set says nothing about cause: a row crosses because it is expensive,
// because the run was slow, or both. Joining the sets across runs decides it. An identity that
// crosses in many runs is expensive and tightening the line reaches it; identities that each cross
// once and never again are being selected by the run, and the line is adjudicating something the
// tree does not control.
//
// IT DOES NOT DECIDE THE QUESTION ON ITS OWN, and the confound is not hypothetical. Runs sampled
// across different tree vintages have had crossing families REPAIRED between them, so an identity
// that stops recurring may have been fixed rather than never re-selected. A recurrence census is
// evidence about cause only over runs whose relevant population did not change, and a caller
// spanning vintages owes that qualification with the result.
// SORTED-SCAN GROUPING, NOT A NESTED FILTER. The first version folded the crossing ids and
// re-filtered the whole list for every unseen one, which is quadratic in the number of crossings,
// and appended to an accumulator with `concat`, which copies. Both are named in DESIGN section 6's
// bare-minimum-cost rule, whose whole point is that "n is small here" is not a time-stable fact:
// the population is per-run crossers today because the floor is healthy, and a bad week is exactly
// when this function gets called on a large one.
//
// So the ids are sorted once and scanned once, runs of equal ids are counted in a single pass, and
// the accumulator is built by prepending and reversed at the end. The output order becomes
// identity order rather than first-appearance order, which is a stronger property than the one it
// replaces -- a census whose order depends on which run happened to be sampled first is not a
// census anyone can diff.
type RecurrenceScan {
started: Bool
current: String
seen: Int
done: List<CrossingRecurrence>
}

fn crossing_recurrence(
runs: List<RunCost>,
ceiling_ms: Millisecond,
excluded_modules: List<String>
) -> List<CrossingRecurrence> {
let crossers = runs
|> flat_map(run => run_crossings(run: run, ceiling_ms: ceiling_ms, excluded_modules: excluded_modules))
|> sort_by(id => id)
let scanned = fold(
crossers,
init: RecurrenceScan { started: false, current: "", seen: 0, done: [] },
f: (acc, id) =>
if acc.started && acc.current == id {
RecurrenceScan { started: true, current: acc.current, seen: acc.seen + 1, done: acc.done }
} else if acc.started {
RecurrenceScan {
started: true,
current: id,
seen: 1,
done: concat([CrossingRecurrence { identity: acc.current, runs_crossed: acc.seen }], acc.done)
}
} else {
RecurrenceScan { started: true, current: id, seen: 1, done: [] }
}
)
if scanned.started {
reverse(concat([CrossingRecurrence { identity: scanned.current, runs_crossed: scanned.seen }], scanned.done))
} else {
[]
}
}

// THE POPULATION THAT DECIDES WHETHER A CEILING IS ADJUDICATING THE TREE OR THE RUN: identities
// observed crossing in more than one of the sampled runs. Empty means every crossing was
// single-run, which is the shape a run-level cause produces.
fn recurring_crossers(recurrence: List<CrossingRecurrence>) -> List<CrossingRecurrence> {
recurrence |> filter(r => r.runs_crossed > 1)
}
Loading
Loading