diff --git a/.changeset/score-floor-and-reach.md b/.changeset/score-floor-and-reach.md new file mode 100644 index 000000000..cf4bbbe4c --- /dev/null +++ b/.changeset/score-floor-and-reach.md @@ -0,0 +1,25 @@ +--- +'@svelte-vitals/core': minor +'svelte-vitals': minor +'@svelte-vitals/vite': minor +--- + +A less severe finding now costs less than a more severe one **within the same (category, scope) pair**, and +the report says how much of a project each category touched. + +A key's category score is the share of that category's severity weight that survived, checks grouped by +category and scope — the keys of the new `inventories` map, like `seo::route`. **Within one pair** a +`warning` costs five times an `info` and a `critical` fifteen times, so a more severe finding always costs +more, there. **Across pairs it does not**: a pair that checks very little is scored against a floor of 25, so +a `warning` there can cost more than a `critical` in a large pair — a `warning` in a floored pair costs 20 +while a `critical` in `seo::route` costs 13.64. A key is now never scored against less than 25 points of +checks: in a one-rule pair the three severities give **96** (`info`), **80** (`warning`) and **40** +(`critical`), where a lone `warning` used to score **0**. + +Scores rise wherever a category checks few things. **A `--min-health` gate calibrated on the previous release +will pass more easily; recalibrate it.** + +Because a score is a mean over every key, forty affected keys and one affected key can display alike. Each +category in the JSON report now carries `keys` and `affectedKeys`, which distinguish them exactly, and +an `inventories` map giving the divisor behind every key of a pair, so a route's per-category score can be +checked by hand (a route's own `score`, which can span more than one pair, cannot). diff --git a/docs/src/content/docs/guides/(reporting)/health-report.md b/docs/src/content/docs/guides/(reporting)/health-report.md index 819cd63f2..0d4772e4d 100644 --- a/docs/src/content/docs/guides/(reporting)/health-report.md +++ b/docs/src/content/docs/guides/(reporting)/health-report.md @@ -16,7 +16,11 @@ Health is computed in two stages: For each active category (SEO, Performance), svelte-vitals computes an independent score: - Each route scores the share of that category's checks it was measured against — weighted by severity — - that passed: no failures scores **100**, every applicable check failing scores **0**. + that passed: no failures scores **100**. A route is never scored against less than 25 points of severity + weight (the _inventory floor_ — see the [Reporters guide](/guides/reporters) for the full rule). The floor + only helps a thin inventory: it stops one or two findings from zeroing out a route that checks very little. + Once that inventory reaches 25 on its own, the floor changes nothing, and a route failing every applicable + check still scores **0**. - Severity sets the weight a failing check carries: `critical` weighs 15, `warning` weighs 5, `info` weighs 1. - A failing check counts once per (route, rule) pair — duplicates take the maximum severity, not a sum. - Route scores are averaged to produce the category's headline score. diff --git a/docs/src/content/docs/guides/(reporting)/reporters.md b/docs/src/content/docs/guides/(reporting)/reporters.md index 8f88c36ce..731691612 100644 --- a/docs/src/content/docs/guides/(reporting)/reporters.md +++ b/docs/src/content/docs/guides/(reporting)/reporters.md @@ -39,7 +39,9 @@ svelte-vitals --reporter json "routeAverage": 94, // mean of the per-route scores, floored "sitePenalty": 0, // deducted for site-wide findings (no route) "criticalCap": null // the cap value when a critical finding lowered the score, else null - } + }, + "keys": 42, // routes (or other scored units) this category measured + "affectedKeys": 6 // of those, how many carried at least one finding } }, "summary": { "critical": 0, "warning": 33, "info": 44, "passed": 610, "dynamic": 2 }, @@ -71,10 +73,33 @@ svelte-vitals --reporter json ] } ], - "siteIssues": [] // findings with no route (robots.txt, sitemap.xml, …), same issue shape + "siteIssues": [], // findings with no route (robots.txt, sitemap.xml, …), same issue shape + "inventories": { + "seo::route": 110 // floored severity weight behind every "seo" key scored against "route" + } } ``` +A category's score on a key is the share of that category's severity weight that survived. Checks are +grouped by category and scope — the keys of `inventories`, like `seo::route` — and **within one group** a +`warning` costs five times an `info` and a `critical` fifteen times, so a more severe finding always costs +more. **Across groups it does not**: a group that checks very few things is scored against a floor of 25, +which makes each of its findings a larger share, so a `warning` in a small group can cost more than a +`critical` in a large one. Repeated findings from the same rule on the same key cost what one costs. Beside +the score, `affectedKeys` says how much of the project the category touched: the score is depth, that is +reach. + +Two things follow that the paragraph above doesn't say directly: + +- per-key scores are comparable **within** a category; across categories the number says which category has + a larger share of _its own_ checks failing, not which problem is worse. +- `inventories` gives the divisor behind every key of one pair, so a route's per-category score + (`routes[].categories`) recomputes by hand from it — this holds because a key is either a route id or a + source file path, and those two key spaces never overlap, so a category's results on one key always share + one scope. A route's own `score` does not recompute the same way, once the route spans more than one pair: + it sums the raw inventory of every pair touched and floors that sum once, while `inventories` publishes + each pair already floored on its own — the two can disagree. + Two field names are worth pointing out, because guessing them wrongly fails silently: - the rule identifier is **`id`**, not `rule`; diff --git a/docs/src/content/docs/ja/guides/(reporting)/health-report.md b/docs/src/content/docs/ja/guides/(reporting)/health-report.md index 9bfacc82f..345d2d727 100644 --- a/docs/src/content/docs/ja/guides/(reporting)/health-report.md +++ b/docs/src/content/docs/ja/guides/(reporting)/health-report.md @@ -16,7 +16,10 @@ Health は 2 段階で計算されます: svelte-vitals は、アクティブなカテゴリ(SEO、パフォーマンスなど)ごとに独立したスコアを計算します: - 各ルートは、そのカテゴリで測定対象となったチェックのうち、重大度で重み付けした上で合格した割合をスコアとします。 - 失敗が一つも無ければ **100**、該当するチェックがすべて失敗すれば **0** になります。 + 失敗が一つも無ければ **100** です。ルートが測定される重大度ウェイトには 25 点の下限値があります + (詳しくは[レポーターガイド](/ja/guides/reporters)を参照)。この下限値が効くのは測定対象がもともと薄いルート + だけで、一つか二つの検出でスコアがゼロになるのを防ぎます。測定対象がそれ自体で 25 点以上あるルートでは下限値は + 何も変えず、該当するチェックがすべて失敗すればスコアは **0** のままです。 - 失敗したチェックの重みは重大度が決めます:`critical` は 15、`warning` は 5、`info` は 1 です。 - 失敗したチェックは(ルート、ルール)ペアごとに一度だけ数えます。同じペアで重複した場合は、合計せず最大の重大度を適用します。 - ルートスコアを平均してカテゴリの見出しスコアを算出します。 diff --git a/docs/src/content/docs/ja/guides/(reporting)/reporters.md b/docs/src/content/docs/ja/guides/(reporting)/reporters.md index a8c6542df..6b9a4ec5a 100644 --- a/docs/src/content/docs/ja/guides/(reporting)/reporters.md +++ b/docs/src/content/docs/ja/guides/(reporting)/reporters.md @@ -39,7 +39,9 @@ svelte-vitals --reporter json "routeAverage": 94, // ルートごとのスコアの平均(切り捨て) "sitePenalty": 0, // サイト全体の検出(route を持たないもの)による減点 "criticalCap": null // critical によってスコアが抑えられた場合はその上限値、なければ null - } + }, + "keys": 42, // このカテゴリが測定した対象(ルートなど)の数 + "affectedKeys": 6 // そのうち何らかの検出があった数 } }, "summary": { "critical": 0, "warning": 33, "info": 44, "passed": 610, "dynamic": 2 }, @@ -71,10 +73,20 @@ svelte-vitals --reporter json ] } ], - "siteIssues": [] // route を持たない検出(robots.txt、sitemap.xml など)。issue の構造は同じ + "siteIssues": [], // route を持たない検出(robots.txt、sitemap.xml など)。issue の構造は同じ + "inventories": { + "seo::route": 110 // "seo" と "route" の組み合わせで測定されるキーそれぞれの、下限値適用後の重大度ウェイト合計 + } } ``` +あるキーに対するカテゴリのスコアは、そのカテゴリが持つ重大度ウェイトのうち生き残った割合です。チェックはカテゴリとスコープの組——`inventories` のキーである `seo::route` のような単位——でグループ化されており、同じ組の中でなら `warning` は `info` の5倍、`critical` は15倍のコストがかかるので、重大度の高い検出は必ずより大きなコストになります。ところが組をまたぐとこの順序は成り立ちません。チェック対象が極端に少ない組は下限値25を分母にスコアが計算されるため、そこでは1件あたりの負担が相対的に大きくなり、小さな組の `warning` が大きな組の `critical` より高くつくことがあります。同じルールが同じキーで何度検出されても、コストは1件分のままです。スコアの隣にある `affectedKeys` は、このカテゴリがプロジェクトのどれだけに触れたかを示します——スコアが深さなら、こちらは到達範囲です。 + +この段落が伝えていないことがもう二つあります。 + +- キーごとのスコアは**同一カテゴリ内でのみ**比較可能です。カテゴリをまたいだ数値が示すのは、どちらの問題がより深刻かではなく、どちらのカテゴリで自身のチェックがより多く失敗しているかです。 +- `inventories` は一つの組に属するすべてのキーの分母を示すので、ルートごとのカテゴリスコア(`routes[].categories`)はそこから手計算で検算できます。これが成り立つのは、キーがルート ID かソースファイルパスのどちらか一方であり、この二つの空間が重ならないため、あるキーの1カテゴリ内の結果が必ず単一のスコープに属するからです。ルート自身の `score` はそうはいきません。複数の組にまたがるルートでは、触れたすべての組の生の重みを合計してから一度だけ下限値を適用するのに対し、`inventories` は各組を個別に下限値適用済みで公開しているため、両者は食い違うことがあります。 + 次の2つのフィールド名は、取り違えても**エラーにならず静かに空振りする**ため、特に注意してください。 - ルールの識別子は **`id`** です(`rule` ではありません) diff --git a/docs/superpowers/plans/2026-08-05-score-floor-and-reach.md b/docs/superpowers/plans/2026-08-05-score-floor-and-reach.md new file mode 100644 index 000000000..e29509679 --- /dev/null +++ b/docs/superpowers/plans/2026-08-05-score-floor-and-reach.md @@ -0,0 +1,573 @@ +# Score floor and reach Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Make a less severe finding cost less than a more severe one, and put the magnitude signal where a +mean cannot carry it. + +**Architecture:** A floor of 25 joins the existing `max(observedInventory, failedWeight)`, so a category that +checks very few things is scored against the floor rather than against those few. `ScoreResult` gains the +counts of keys touched and keys penalized, which the JSON report exposes per category. The report also gains +the floored weight of every `(category, scope)` pair, so the arithmetic is checkable. + +**Tech Stack:** TypeScript (ESM, `.js` import specifiers), vitest, oxlint + oxfmt, Astro Starlight docs. + +**Spec:** `docs/superpowers/specs/2026-08-05-score-floor-and-reach-design.md`. **Read it before Task 1.** Its +predecessor was withdrawn after field review, and its own numbers were measured rather than derived — if one +disagrees with what you compute, stop and report it rather than adjusting a test. + +## Global Constraints + +- **`packages/core/src/` is runtime-agnostic**: no `node:` imports, no I/O, no runtime-specific globals. +- **The floor is `25`, one named constant beside `CRITICAL_CAP`.** `inventoryWeight = max(observedInventory, failedWeight, 25)`. +- **`DEDUCTION` values, `CRITICAL_CAP`, `Math.floor`, the deficit-space mean, and `sitePenalty` do not + change.** `sitePenalty` stays absolute points subtracted after the mean. +- **The invariant survives**: no penalized finding → 100; one `info` among many passes → never 100. +- **Comments and docs are for the next reader** (`AGENTS.md`): a line earns its place only when it says + something the code cannot. Prefer one line over three. **Test names state the behaviour, not the reasoning.** +- **Never name another tool, linter, plugin or product** in any doc, comment or commit message. +- **en/ja docs ship together** — the Japanese must be idiomatic, not a transliteration. +- Conventional commits, scoped by package. **A changeset is required** — `minor`, listing + `@svelte-vitals/core`, `svelte-vitals`, `@svelte-vitals/vite`. + +## File Structure + +| File | Responsibility | +| -------------------------------------------------------------- | ------------------------------------------------------------------------ | +| `packages/core/src/scoring/score.ts` | the floor constant, the `max`, and the two new `ScoreResult` counts. | +| `packages/core/src/reporter/json.ts` | expose `keys` and `affectedKeys` per category, and the pair inventories. | +| `packages/core/test/score.test.ts` | the floor, the ordering, the counts. | +| `packages/core/test/json-report.test.ts` | the report shape. | +| `packages/core/test/*.test.ts`, `packages/vite/test/*.test.ts` | expectations that move because scores rise. | +| `docs/src/content/docs/guides/(reporting)/reporters.md` + ja | the paragraph, the sample, the three facts. | +| `.changeset/score-floor-and-reach.md` | **new** | + +Three tasks. The floor changes every score, so it lands alone and its fallout is re-baselined with it; the +counts are additive and separable; the docs are their own reviewable deliverable. + +--- + +## Task 1: The floor + +**Files:** + +- Modify: `packages/core/src/scoring/score.ts` (the `DEDUCTION`/`CRITICAL_CAP` block near line 5, and the + `inventoryWeight` line inside `computeScore`) +- Test: `packages/core/test/score.test.ts` (append) + +**Interfaces:** + +- Consumes: nothing new. +- Produces: `export const INVENTORY_FLOOR = 25;` — Task 2 does not use it, Task 3 imports it so the report + and the scorer cannot drift. + +- [ ] **Step 1: Write the failing tests** + +Append to `packages/core/test/score.test.ts`. The `fail`, `pass` and `r` helpers already exist in that file — +read the top of it and reuse them rather than redefining. + +```ts +describe('computeScore — inventory floor', () => { + const config = defineConfig({}); + + it('scores a one-rule pair at 80 for a single warning', () => { + // Inventory 5 floored to 25: 100 − 500/25. + const rules = [r('x/only', 'seo', 'component', 'warning')]; + const results = [fail('x/only', 'src/A.svelte', 'warning')]; + expect(computeScore(results, config, { rules, applyCriticalCap: false }).score).toBe(80); + }); + + it('scores an eight-info pair at 96 for a single info', () => { + // Inventory 8 floored to 25: 100 − 100/25. + const rules = Array.from({ length: 8 }, (_, i) => r(`a/${i}`, 'architecture', 'component', 'info')); + const results = [fail('a/0', 'src/A.svelte', 'info')]; + expect(computeScore(results, config, { rules }).score).toBe(96); + }); + + it('leaves a pair above the floor unchanged', () => { + // Inventory 30 > 25, so one warning costs 500/30 and the key scores 83. + const rules = Array.from({ length: 6 }, (_, i) => r(`s/${i}`, 'seo', 'route', 'warning')); + const results = [fail('s/0', '/a', 'warning')]; + expect(computeScore(results, config, { rules }).score).toBe(83); + }); + + it('still scores 100 when nothing is penalized', () => { + const rules = [r('x/only', 'seo', 'component', 'warning')]; + expect(computeScore([pass('x/only', 'src/A.svelte')], config, { rules }).score).toBe(100); + }); + + it('still subtracts sitePenalty in absolute points', () => { + // A site-wide warning costs 5, not a share of anything — the floor must not reach it. + const rules = [r('p/site', 'seo', 'project', 'warning'), r('s/one', 'seo', 'route', 'warning')]; + const results = [ + { + id: 'p/site', + category: 'seo', + severity: 'warning', + detection: { presence: 'none', value: 'absent' }, + message: 'x' + }, + pass('s/one', '/a') + ] as Result[]; + const sr = computeScore(results, config, { rules }); + expect(sr.scoreModel.sitePenalty).toBe(5); + expect(sr.score).toBe(95); + }); + + it('orders info below warning in every registry pair', () => { + // Derived from the registry rather than written out, so a new rule cannot silently break the ordering. + const inv = buildInventory(config); + const cost = (weight: number, i: number) => (100 * weight) / Math.max(i, 25); + const worstInfo = Math.max(...[...inv.values()].map((i) => cost(1, i))); + const cheapestWarning = Math.min(...[...inv.values()].map((i) => cost(5, i))); + expect(worstInfo).toBeLessThanOrEqual(cheapestWarning); + }); +}); +``` + +Add `buildInventory` to that file's existing import from `../src/scoring/inventory.js` if it is not already +imported. + +- [ ] **Step 2: Run the tests to verify they fail** + +Run from `packages/core`: `../../node_modules/.bin/vitest run test/score.test.ts -t 'inventory floor'` +Expected: FAIL — the first case scores 0, the second 87, and the ordering assertion is false. The two +"still" cases pass already; they are there to fail if the floor reaches something it must not. + +- [ ] **Step 3: Add the floor** + +In `packages/core/src/scoring/score.ts`, beside `CRITICAL_CAP`: + +```ts +/** + * A key is never scored against less than this much severity weight. Without it a pair holding one rule + * makes that rule's finding cost the whole key, and a finding's cost stops tracking its severity — an + * `info` in an eight-rule pair outweighed a `warning` in a twenty-six-rule one. + */ +export const INVENTORY_FLOOR = 25; +``` + +and change the `inventoryWeight` line inside `computeScore` from + +```ts +inventoryWeight = Math.max(inventoryWeight, failed); +``` + +to + +```ts +inventoryWeight = Math.max(inventoryWeight, failed, INVENTORY_FLOOR); +``` + +Leave the comment above that line in place and extend it with one clause naming the floor's job; do not +replace what it says about `failed`. + +- [ ] **Step 4: Run the tests to verify they pass** + +Run: `../../node_modules/.bin/vitest run test/score.test.ts -t 'inventory floor'` +Expected: PASS, 6 tests. + +- [ ] **Step 5: Prove the floor is load-bearing** + +Remove `INVENTORY_FLOOR` from the `Math.max`, confirm the first, second and ordering cases fail and the two +"still" cases stay green, restore, confirm all six green. Report both halves — a floor that also broke the +unchanged cases would be reaching too far. + +- [ ] **Step 6: Re-baseline the expectations that move** + +```bash +for p in core cli vite; do (cd packages/$p && ../../node_modules/.bin/vitest run); done +``` + +Every score against a pair below 25 rises. **Recompute each failing expectation from the formula** +`100 − (100 × failedWeight) / max(observedInventory, failedWeight, 25)` — never paste the number the runner +prints, which makes the test agree with the implementation instead of the design. Get real inventories from +`buildInventory(config)` in a scratch check and delete the scratch file. + +**A failure that is not a score or health number is a real regression — stop and report it** rather than +editing the expectation. + +- [ ] **Step 7: Typecheck and lint** + +```bash +(cd packages/core && ../../node_modules/.bin/tsup) +for p in core cli vite; do (cd packages/$p && ../../node_modules/.bin/tsc --noEmit); done +node_modules/.bin/oxlint . && node_modules/.bin/oxfmt --check . +``` + +Expected: clean. (`packages/mcp` has no `tsconfig.json`; skip it.) + +- [ ] **Step 8: Commit** + +```bash +git add packages/core/src/scoring/score.ts packages/core/test packages/cli/test packages/vite/test +git commit -m "fix(core): never score a key against less than 25 points of checks" +``` + +--- + +## Task 2: Reach + +**Files:** + +- Modify: `packages/core/src/scoring/score.ts` (`ScoreResult`, and the return in `computeScore`) +- Modify: `packages/core/src/reporter/json.ts` (`JsonReport['categories']`, and the `categories` assembly) +- Test: `packages/core/test/score.test.ts`, `packages/core/test/json-report.test.ts` + +**Interfaces:** + +- Consumes: `computeScore` as changed in Task 1. +- Produces: + + ```ts + export interface ScoreResult { + score: number; + rawScore: number; + scoreModel: ScoreModel; + /** Keys this result set touched. */ + keys: number; + /** Keys carrying at least one penalized finding. */ + affectedKeys: number; + } + ``` + + and `JsonReport['categories']` becomes + `Record`. + +- [ ] **Step 1: Write the failing tests** + +Append to `packages/core/test/score.test.ts`: + +```ts +describe('computeScore — reach', () => { + const config = defineConfig({}); + const rules = Array.from({ length: 8 }, (_, i) => r(`a/${i}`, 'architecture', 'component', 'info')); + + it('counts keys touched and keys penalized', () => { + const results = [fail('a/0', 'src/A.svelte', 'info'), pass('a/0', 'src/B.svelte'), pass('a/0', 'src/C.svelte')]; + const sr = computeScore(results, config, { rules }); + expect(sr.keys).toBe(3); + expect(sr.affectedKeys).toBe(1); + }); + + it('reports the same score and different reach for one finding and for many', () => { + // The reason reach exists: after the floor these two score alike and must still be distinguishable. + const keys = Array.from({ length: 40 }, (_, i) => `src/${i}.svelte`); + const one = keys.map((k, i) => (i === 0 ? fail('a/0', k, 'info') : pass('a/0', k))); + const many = keys.map((k) => fail('a/0', k, 'info')); + const a = computeScore(one, config, { rules }); + const b = computeScore(many, config, { rules }); + expect(a.score).toBe(b.score); + expect(a.affectedKeys).toBe(1); + expect(b.affectedKeys).toBe(40); + }); + + it('counts a key once however many rules penalize it', () => { + const results = [fail('a/0', 'src/A.svelte', 'info'), fail('a/1', 'src/A.svelte', 'info')]; + expect(computeScore(results, config, { rules }).affectedKeys).toBe(1); + }); +}); +``` + +Append to `packages/core/test/json-report.test.ts`, reusing that file's module-level `config` and declaring +each case's results inline: + +```ts +describe('buildJsonReport — category reach', () => { + it('reports keys and affectedKeys per category', () => { + const results: Result[] = [ + { + id: 'seo/canonical-url', + category: 'seo', + severity: 'warning', + detection: { presence: 'none', value: 'absent' }, + route: '/a', + message: 'x' + }, + { + id: 'seo/canonical-url', + category: 'seo', + severity: 'warning', + detection: { presence: 'own', value: 'static' }, + route: '/b', + message: 'ok' + } + ]; + const report = buildJsonReport(results, config, { version: '0.0.0' }); + expect(report.categories.seo!.keys).toBe(2); + expect(report.categories.seo!.affectedKeys).toBe(1); + }); +}); +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run from `packages/core`: +`../../node_modules/.bin/vitest run test/score.test.ts test/json-report.test.ts -t 'reach'` +Expected: FAIL — `keys` and `affectedKeys` are `undefined`. + +- [ ] **Step 3: Count and return** + +In `computeScore`, the loop over `observed` already visits every key. Count there: + +```ts + let affectedKeys = 0; + let totalDeficit = 0; + for (const [key, pairs] of observed) { + let failed = 0; + for (const d of ruleMax.get(key)?.values() ?? []) failed += d; + if (failed > 0) affectedKeys += 1; +``` + +and extend the return: + +```ts +return { + score: Math.floor(rawScore), + rawScore, + scoreModel: { routeAverage, sitePenalty, criticalCap }, + keys: keyCount, + affectedKeys +}; +``` + +`keyCount` is already computed as `observed.size`. Add the two fields to `ScoreResult` with the docstrings +from the Interfaces block above. + +- [ ] **Step 4: Expose them in the report** + +In `packages/core/src/reporter/json.ts`, widen the interface line: + +```ts +categories: Record; +``` + +and the assembly: + +```ts +const categories = Object.fromEntries( + Object.entries(byCat).map(([cat, sr]) => [ + cat, + { score: sr.score, scoreModel: sr.scoreModel, keys: sr.keys, affectedKeys: sr.affectedKeys } + ]) +); +``` + +- [ ] **Step 5: Run the tests to verify they pass** + +Run: `../../node_modules/.bin/vitest run test/score.test.ts test/json-report.test.ts -t 'reach'` +Expected: PASS, 4 tests. + +- [ ] **Step 6: Fix what stops compiling** + +`keys` and `affectedKeys` are required on `ScoreResult` and on the report's `categories`, so hand-built +literals in test fixtures break. A text search for the type name will not find them all — several are typed +through another type: + +```bash +(cd packages/core && ../../node_modules/.bin/tsup) +for p in core cli vite; do (cd packages/$p && ../../node_modules/.bin/tsc --noEmit); done +``` + +Give each fixture `keys: 0, affectedKeys: 0` — these exercise rendering, not scoring, and zero is the honest +value for a literal with no scored results. Do not invent counts. + +- [ ] **Step 7: Run everything** + +```bash +for p in core cli vite; do (cd packages/$p && ../../node_modules/.bin/vitest run); done +(cd packages/core && ../../node_modules/.bin/tsup) +for p in core cli vite; do (cd packages/$p && ../../node_modules/.bin/tsc --noEmit); done +node_modules/.bin/oxlint . && node_modules/.bin/oxfmt --check . +``` + +Expected: all green. + +- [ ] **Step 8: Commit** + +```bash +git add packages/core/src packages/core/test packages/vite/test +git commit -m "feat(core): report how many keys a category reached" +``` + +--- + +## Task 3: The inventories, and the documentation + +**Files:** + +- Modify: `packages/core/src/reporter/json.ts` (`JsonReport`, and `buildJsonReport`) +- Test: `packages/core/test/json-report.test.ts` +- Modify: `docs/src/content/docs/guides/(reporting)/reporters.md` and + `docs/src/content/docs/ja/guides/(reporting)/reporters.md` +- Create: `.changeset/score-floor-and-reach.md` + +**Interfaces:** + +- Consumes: Tasks 1 and 2, and `buildInventory` / `pairKey` from `packages/core/src/scoring/inventory.js`. +- Produces: `JsonReport` gains `inventories: Record`, keyed `"::"`, holding + the **floored** weight each key of that pair is scored against. + +**Why here and not on `scoreModel`.** A `ScoreModel` describes one `computeScore` call, and the call behind a +category covers many keys of possibly different pairs — there is no single inventory weight to report there, +and reporting one key's would be a number that explains nothing. The pair map is unambiguous, is the same for +every key of a pair, and is what a reader needs to check any of them. + +- [ ] **Step 1: Write the failing test** + +Append to `packages/core/test/json-report.test.ts`, reusing that file's module-level `config`: + +```ts +describe('buildJsonReport — pair inventories', () => { + it('reports the floored weight of each pair', () => { + const results: Result[] = [ + { + id: 'seo/canonical-url', + category: 'seo', + severity: 'warning', + detection: { presence: 'none', value: 'absent' }, + route: '/a', + message: 'x' + } + ]; + const report = buildJsonReport(results, config, { version: '0.0.0' }); + // seo::route holds 110 points and is above the floor; architecture::component holds 8 and is not. + expect(report.inventories['seo::route']).toBe(110); + expect(report.inventories['architecture::component']).toBe(25); + }); + + it('lets a reader recompute a route category score from the map', () => { + const results: Result[] = [ + { + id: 'seo/canonical-url', + category: 'seo', + severity: 'warning', + detection: { presence: 'none', value: 'absent' }, + route: '/a', + message: 'x' + } + ]; + const report = buildJsonReport(results, config, { version: '0.0.0' }); + const i = report.inventories['seo::route']!; + expect(Math.floor(100 - (100 * 5) / i)).toBe(report.routes[0]!.categories.seo); + }); +}); +``` + +- [ ] **Step 2: Run the test to verify it fails** + +Run from `packages/core`: `../../node_modules/.bin/vitest run test/json-report.test.ts -t 'pair inventories'` +Expected: FAIL — `report.inventories` is `undefined`. + +- [ ] **Step 3: Build the map** + +In `packages/core/src/reporter/json.ts`, add to the interface: + +```ts +/** Floored severity weight per `"::"` pair — the divisor behind every key of that pair. */ +inventories: Record; +``` + +and in `buildJsonReport`, beside the other derived values: + +```ts +const inventories = Object.fromEntries( + [...buildInventory(config)].map(([pair, weight]) => [pair, Math.max(weight, 25)]) +); +``` + +Import `buildInventory` from `../scoring/inventory.js`. Add `inventories` to the returned object. + +**The `25` here must not be a second literal.** Export `INVENTORY_FLOOR` from +`packages/core/src/scoring/score.ts` (Task 1 declared it) and use it, so the report and the scorer cannot +drift. + +- [ ] **Step 4: Run the test to verify it passes** + +Run: `../../node_modules/.bin/vitest run test/json-report.test.ts -t 'pair inventories'` +Expected: PASS, 2 tests. Then run the full core suite; `JsonReport` literals in +`packages/core/test/html-report.test.ts` and the two `packages/vite` fixtures need `inventories: {}`. + +- [ ] **Step 5: Write the guide paragraph** + +`docs/src/content/docs/guides/(reporting)/reporters.md` documents the report shape. Add the sample fields +(`keys`, `affectedKeys`, and the `inventories` map) and this paragraph, adapted to the page's voice: + +> A category's score on a key is the share of that category's severity weight that survived. One `info` costs +> a twenty-fifth of the weight at most, one `warning` five times that, one `critical` fifteen times — so a +> more severe finding always costs more than a less severe one within a category, and a category that checks +> very few things is scored against a floor rather than against those few. Repeated findings from the same +> rule on the same key cost what one costs. Beside the score, `affectedKeys` says how much of the project the +> category touched: the score is depth, that is reach. + +Then add the two facts the paragraph does not carry: + +- per-key scores are comparable **within** a category; across categories the number says which category has a + larger share of _its own_ checks failing, not which problem is worse; +- `inventories` gives the divisor behind every key of a pair, so any score can be recomputed by hand. + +Mirror all of it in the Japanese page. **Do not transliterate** — write it as Japanese technical prose. + +- [ ] **Step 6: Write the changeset** + +Create `.changeset/score-floor-and-reach.md`: + +```md +--- +'@svelte-vitals/core': minor +'svelte-vitals': minor +'@svelte-vitals/vite': minor +--- + +A less severe finding now costs less than a more severe one, and the report says how much of a project each +category touched. + +A key's category score is the share of that category's severity weight that survived, so a finding's cost +depends on how much that category checks. Where a category checked very little, a single `info` could cost +more than a `warning` elsewhere — measured on a real project, an `info` took 13 points off a key while a +`warning` took 5 — and a category holding one rule scored a key **0** for one finding. A key is now never +scored against less than 25 points of checks, which orders `info` below `warning` everywhere and turns that +0 into 80. + +Scores rise wherever a category checks few things. **A `--min-health` gate calibrated on the previous release +will pass more easily; recalibrate it.** + +Because a score is a mean over every key, forty affected keys and one affected key can display alike. Each +category in the JSON report now carries `keys` and `affectedKeys`, which distinguish them exactly, and +an `inventories` map giving the divisor behind every key, so the arithmetic can be checked. +``` + +- [ ] **Step 7: Run everything** + +```bash +for p in core cli vite; do (cd packages/$p && ../../node_modules/.bin/vitest run); done +(cd packages/core && ../../node_modules/.bin/tsup) +for p in core cli vite; do (cd packages/$p && ../../node_modules/.bin/tsc --noEmit); done +node_modules/.bin/oxlint . && node_modules/.bin/oxfmt --check . +(cd packages/cli && ../../node_modules/.bin/vitest run test/docs-links.test.ts test/rules-index.test.mjs test/docs-embed.test.mjs) +node scripts/floor-smoke.mjs +``` + +Expected: all green. + +- [ ] **Step 8: Commit** + +```bash +git add packages/core docs .changeset +git commit -m "feat(core): expose the weight each pair is scored against" +``` + +--- + +## Notes for whoever runs this + +- A full-workspace `pnpm` command fails in this sandbox for a known, pre-existing reason (the `docs` package's + dependencies), and `pnpm --filter build` fails on a reflink error. Use `../../node_modules/.bin/tsup` + from inside the package. +- The spec records four things as deliberately out of scope: ordering `warning` below `critical` across + categories (no floor achieves it), excluding unconfigured rules from the denominator (the floor subsumes + it), fixing dilution (reach is the answer instead), and rendering reach in the HTML report or dashboard. +- The predecessor design was withdrawn after field review found its central justification false. If you are + about to write a comment or a doc sentence describing _why_ the model behaves as it does, take the wording + from the spec rather than paraphrasing — the last branch shipped a flat claim into a code comment that the + spec had spent four review passes making precise. diff --git a/docs/superpowers/specs/2026-08-05-score-floor-and-reach-design.md b/docs/superpowers/specs/2026-08-05-score-floor-and-reach-design.md new file mode 100644 index 000000000..831d8a6a7 --- /dev/null +++ b/docs/superpowers/specs/2026-08-05-score-floor-and-reach-design.md @@ -0,0 +1,152 @@ +# A floor under the denominator, and reach beside the score — design + +**Date:** 2026-08-05 +**Status:** approved +**Supersedes:** `2026-08-05-score-semantics-design.md`, withdrawn after field review. +**Origin:** a field measurement on a real project (351 keys, 77 findings) and the review of the withdrawn +design, both 2026-08-05. + +## The problem, in the order it has to be solved + +**A finding labelled less severe costs more.** On the field project an `architecture` `info` takes **13 +points** off a key while a `seo` `warning` takes **5**. That is 41 of 351 keys, not a corner case. The +severity a rule declares and the damage it does have come apart. + +**The denominator counts checks that never ran.** `architecture::component` holds 8 rules, of which **6 +evaluate nothing** — verified on the repo's own fixture, where `.rules` reports `0 findings / 0 passed` for +six of the eight. The withdrawn design justified the model as "the share of what we check here that failed"; +what is actually checked there is two things, not eight. + +**The two are the same knob.** In a ratio model a finding costs `1 / inventory`. Making the denominator +honest shrinks it, which makes each finding cost _more_ — measured on the field project, dropping +`architecture::component` from 8 to the 4 configured rules moves an `info` from 13 points to 25, and to the 2 +evaluated rules moves it to 50. Every fix for the honesty problem makes the severity problem worse, and the +obvious fix for the severity problem — a larger denominator — makes the score move less. The withdrawn design +tried to settle the second without the first and had to refuse both. + +## The design + +### 1. Floor the denominator at 25 + +```text +inventoryWeight = max(observedInventory, failedWeight, 25) +``` + +The existing `max(observedInventory, failedWeight)` stays; 25 joins it. + +**This orders `info` below `warning` everywhere.** The worst `info` costs `100/25 = 4.00`; the cheapest +`warning` costs `500/110 = 4.55`, in `seo::route`. Every pair, after the floor: + +| pair | inventory | floored to | one `info` | one `warning` | one `critical` | +| ------------------------- | --------- | ---------- | ---------- | ------------- | -------------- | +| `seo::route` | 110 | 110 | 99 | 95 | 86 | +| `correctness::component` | 96 | 96 | 98 | 94 | 84 | +| `security::component` | 35 | 35 | 97 | 85 | 57 | +| `performance::route` | 28 | 28 | 96 | 82 | 46 | +| `seo::project` | 16 | **25** | 96 | 80 | 40 | +| `performance::component` | 9 | **25** | 96 | 80 | 40 | +| `architecture::component` | 8 | **25** | 96 | 80 | 40 | +| `seo::component` | 5 | **25** | 96 | 80 | 40 | +| `performance::project` | 5 | **25** | 96 | 80 | 40 | + +The zero-point cases disappear with it: `seo::component`'s lone `warning` scored **0** and now scores **80**. + +**`warning` is not ordered below `critical`, and no floor achieves that.** A `warning` in a floored pair costs +20; a `critical` in `seo::route` costs 13.64. Ordering those would need the floor at ~37, where seven of the +nine pairs sit on the floor and the model has become absolute deductions wearing a ratio's clothes. So it is +left unordered, deliberately, because the case requires a thin pair carrying a `warning` beside a thick pair +carrying a `critical` — and **no thin pair fired at all in the field**: `seo::component`, +`performance::project` and `performance::component` were 0 keys of 351. + +**The floor also settles the honesty problem, which is why it is the only change here.** Every pair whose +denominator was polluted by never-evaluated rules is a thin pair, and every thin pair now sits on the floor — +so excluding unconfigured rules would produce the identical number. Excluding them was the field's own +suggestion and it is the right instinct; the floor reaches it without making the denominator depend on +configuration, which would have meant a project's existing findings getting lighter as it declared more +conventions. + +### 2. Report reach beside the score + +Each category in the JSON report gains the count of keys it touched and the count it penalized: + +```jsonc +"categories": { + "architecture": { "score": 99, "scoreModel": { … }, "keys": 351, "affectedKeys": 41 } +} +``` + +**This is where magnitude now lives.** The score is a mean over every key, so it is `share × depth` — 41 +affected keys of 351 move a category by less than a point once the floor is in. That is not a defect to be +tuned away; it is what a mean of mostly-clean keys says. Splitting the product's two factors out is what makes +both legible: `41 of 351` distinguishes one finding from forty-one exactly, where the score cannot. + +It also removes the reason the withdrawn design refused to floor. That refusal was to protect the score's +resolution — 29 affected keys per displayed point at inventory 8, 88 at 25. With reach reported, resolution +stops carrying the signal and the floor costs nothing that matters. + +### 3. Say what the number means + +Three facts a reader cannot get from the output today, all of which the field measurement discovered by +observation: + +- **A category score is a proportion of severity weight, not a severity ranking.** Comparable within a + category; across categories it says which category has a larger share of its own checks failing. +- **Repeated findings from one rule cost the same as one.** The deduction is per distinct rule id on a key, + duplicates taking the maximum. A key with eight `correctness/each-index-key` findings scores what a key with + one scores. Stated in `2026-08-04-score-proportionality-design.md`; stated nowhere a user reads. +- **The per-pair inventory is not derivable from the report.** A reader who wants to check `96 = 100 − 100/25` + cannot, because neither the inventory nor the floor appears anywhere. The report gains an `inventories` map + from `"::"` to the floored weight, so the arithmetic is checkable. It goes there rather + than on `scoreModel` because a `ScoreModel` describes one `computeScore` call, and a category's call spans + many keys of possibly different pairs — there is no single weight to report, and reporting one key's would + explain nothing. This recomputes `routes[].categories` exactly, because a key is either a route id or a + source file path — the two key spaces never overlap, so a category's results on one key always draw on a + single scope. It does not recompute a route's own `score`: a route spanning more than one pair sums their + raw weights and floors that sum once, while `inventories` publishes each pair already floored on its own. + +The one-paragraph version, which goes in the reporters guide in both languages: + +> A category's score on a key is the share of that category's severity weight that survived. Checks are +> grouped by category and scope — the keys of `inventories`, like `seo::route` — and **within one group** a +> `warning` costs five times an `info` and a `critical` fifteen times, so a more severe finding always costs +> more. **Across groups it does not**: a group that checks very few things is scored against a floor of 25, +> which makes each of its findings a larger share, so a `warning` in a small group can cost more than a +> `critical` in a large one. Repeated findings from the same rule on the same key cost what one costs. +> Beside the score, `affectedKeys` says how much of the project the category touched: the score is depth, +> that is reach. + +## What this costs + +**Scores rise.** On the field project `architecture` goes from 98 to **99**, and its worst keys — two failing +rules against an inventory of 8 — from **75** to **92**. Every thin-pair key moves up. That is the trade for +severity behaving, and the reach count is what keeps the change from hiding anything: 41 of 351 was invisible +before and is now printed. + +**A `--min-health` gate calibrated on the current release will pass more easily.** Recalibrate. + +## Testing + +1. **`info` costs less than `warning` in every pair.** Assert the ordering across all nine pairs + programmatically from the registry rather than as nine literals, so a new rule cannot silently break it. +2. **The floor binds only below 25.** `seo::route` at 110 is unchanged; `architecture::component` at 8 scores + as if 25. Assert both on the same input. +3. **A lone `warning` in a one-rule pair scores 80, not 0.** This is the case the field could not produce and + the floor exists for. +4. **`affectedKeys` counts keys with at least one penalized result in that category**, and `keys` counts every + key the category touched. A category with one finding and one with forty must differ here even when their + scores do not — that is the whole point of the field, so assert both scores equal and both reaches + different on one input. +5. **`inventories` carries the floored weight of every pair**, so a reader recomputing `100 − 100·f/i` gets + the displayed score. Assert one floored pair and one unfloored one, and assert the recomputation against a + real `routes[].categories` value rather than against the map alone. +6. **The invariant survives.** No penalized finding → 100. One `info` among many passes → never 100. +7. **`sitePenalty` is untouched** — still absolute points, still subtracted after the mean. + +## Deliberately not solved + +- **`warning` below `critical` across categories.** See above: unreachable without collapsing the model, and + unobserved in the field. +- **Excluding unconfigured rules from the denominator.** Subsumed by the floor for every pair where it would + change a number, and rejected on its own because it would make findings lighter as a project declares more. +- **Dilution.** The score is a mean and stays one. Reach is the answer to "how much", not a rounder score. +- **Rendering reach in the HTML report or the dashboard.** They receive the field and ignore it. diff --git a/docs/superpowers/specs/2026-08-05-score-semantics-design.md b/docs/superpowers/specs/2026-08-05-score-semantics-design.md new file mode 100644 index 000000000..cf6676847 --- /dev/null +++ b/docs/superpowers/specs/2026-08-05-score-semantics-design.md @@ -0,0 +1,144 @@ +# What a score means, and what it does not — design + +**Date:** 2026-08-05 +**Status:** withdrawn 2026-08-05 — superseded by `2026-08-05-score-floor-and-reach-design.md`. +Field review rejected its central justification: it described the model as coverage while the denominator +counts rules that evaluated nothing (6 of 8 in `architecture::component`, verified on the repo's own fixture). +It also refused to floor the denominator on grounds that its own table undermined — the magnitude signal it +protected expires once a pair reaches 14 rules, which the project is adding toward anyway. **The measurements +below stand and are the input to the successor**; only the conclusion is withdrawn. +**Origin:** a field measurement of the proportional score model on a real project (351 keys, 77 findings), +reported 2026-08-05. It answered the question +`2026-08-04-score-proportionality-design.md` left open under "severity recalibration". + +## What the field measured + +The measurement asked whether the thin `(category, scope)` pairs — the ones whose small inventories produce +extreme scores — actually fire. Three results, and the surprising one is not the one the question was about. + +**The catastrophic case does not occur.** `seo::component` and `performance::project` each hold exactly one +rule, so one finding scores the key **0**. Neither fired: 0 keys of 351. `performance::component` (inventory 9) did not fire either. The only thin pair that fires is `architecture::component`, on 41 of 351 keys, and +entirely through `architecture/component-size` and `architecture/prop-count`. + +**Severity is inverted across categories, and it is visible.** An `architecture` `info` costs a key **13 +points** (inventory 8); a `seo` `warning` costs **5** (inventory 110). A finding the rule set labels less +severe costs 2.6× more. This is not theoretical — it is what 41 of 351 keys show. + +**The previous release worked.** `architecture` reads 98, having read **100 with the same 43 findings** before +the proportional model shipped. The lie is gone. The dilution is not: 41 affected keys move the category by +two points. + +## The decision: the model does not change + +The obvious response to the inversion is to floor the denominator — score against `max(inventory, K)` so thin +pairs stop charging so much. Measured against the field's own numbers, **there is no K that works.** + +| K | an `architecture` `info` costs | a `seo` `warning` costs | 41 affected keys | 1 affected key | distinguishes? | +| --------- | ------------------------------ | ----------------------- | ---------------- | -------------- | -------------- | +| 8 (today) | 12.5 | 5 | 98 | 99 | **yes** | +| 11 | 9.1 | 5 | 98 | 99 | yes | +| 12 | 8.3 | 5 | 98 | 99 | yes | +| 14 | 7.1 | 5 | **99** | 99 | **no** | +| 20 | 5.0 | 5 | **99** | 99 | **no** | + +The model reproduces the field's reported category score exactly at K = 8, which is what makes the rest of the +table trustworthy: 41 keys deficit `100/8` and 2 keys deficit `200/8` over 351 keys gives 98.47, and the field +reported 98. + +Read the table's two ends. **K ≤ 12 keeps the magnitude signal and keeps the inversion. K ≥ 14 softens the +inversion and destroys the signal** — 41 affected keys and one affected key both display 99. At K = 20, where +an `info` finally costs exactly what a `seo` `warning` costs, the category can no longer tell one finding from +forty-one. + +That is not a tuning failure. **In a coverage model a finding costs `1 / inventory`, so as long as one pair +holds 8 rules and another holds 26, no severity weighting can order them across categories.** The only way to +make severity comparable across pairs is to equalise the inventories — by adding rules, or by flooring — and +flooring shrinks the deduction by exactly the amount it equalises. The two properties are the same knob turned +opposite ways. + +So the choice is which property to keep, and this design keeps the magnitude signal, for three reasons in +descending weight: + +1. **Flooring would undo, one release later, what the previous release shipped and this measurement just + confirmed in the field.** The separation between one finding and forty-one is one point today and zero at + K = 20. Not halved — gone. +2. **The inversion is a correct reading, not an arithmetic error.** `architecture` 87 beside `seo` 95 says + "12.5% of what we check here failed, against 4.5% there", and that is true. The score has never claimed to + rank findings by severity; it reports coverage. What is missing is that nobody wrote that down. +3. **The alarming symptom is theoretical.** A floor would trade a property that occurs on 41 keys for + insurance against one that occurs on none. + +`K = 10` deserves naming because it is the one setting that closes the zero-point cases while keeping the +signal — `seo::component`'s lone `warning` would score 50 rather than 0, and `architecture` would still read +98 against 99. It is not adopted: 50 is nearly as alarming as 0 for a single `warning`, and neither occurs. +Recorded so the option is visible if a zero ever appears in the field. + +## What changes: the documentation + +The measurement's real finding is that the semantics are undocumented. Three things a reader cannot learn from +the output: + +**A score is coverage, not severity.** Per-key scores are comparable **within** a category and not across it. +`architecture 87` next to `seo 95` on the same key does not mean the architecture problem is worse; it means +architecture checks fewer things there, so each one is a larger share. The reporters guide and the health +report guide both describe the score without saying this. + +**Repeated findings from one rule cost the same as one.** The deduction is per distinct rule id on a key, with +duplicates taking the maximum. A key with eight `correctness/each-index-key` findings scores exactly what a key +with one scores. This is deliberate — a rule either passes on a key or does not — and it is stated in +`2026-08-04-score-proportionality-design.md`, but nowhere a user reads. The field measurement discovered it by +observation, which is how a user will too. + +**Magnitude is visible at the category level, not within a key.** "One finding and several hundred now display +differently" holds because more keys become affected, not because a key gets worse as findings accumulate on +it. Both halves need saying together, or the first half reads as a promise the second half breaks. + +## The field corpus, recorded as the third measurement + +`2026-08-04-route-category-scores-design.md` records two corpora for how often `routes[].score` equals the mean +of `routes[].categories`, and says neither predicts a given project. The field is the third, and it lands +between them: + +| corpus | keys compared | agreement | +| ---------------------------- | -------------------------------- | --------- | +| the repo's fixtures | 51, of which 46 single-category | 98% | +| a 200-page synthetic project | 413, of which 400 multi-category | 52% | +| **a real project** | **328 multi-category of 351** | **85%** | + +The real project is multi-category almost everywhere — 263 keys carry two categories, 64 carry three — so its +85% is not the fixtures' single-category artefact. It is high because most keys are clean, which is what the +"clean keys agree by construction" rule predicts and what the synthetic corpus, whose pages are uniformly +flawed, could not show. + +**One correction the field forces.** The design's only worked example has `routes[].score` **below** the mean +(96 against 97.5). The field's deviations run the other way — 30 keys at +1, 10 at +4, 1 at +7, against 8 keys +spread over −1, −3 and −5. The direction has a rule, verified here: with one clean partner category, +`sign(score − mean) = sign(i_clean − i_failing)`. The example has the large-inventory category failing, which +is the minority case; the field's typical key has thin `architecture` failing beside a fat clean partner, which +pushes the union ratio up. A reader calibrating on that example would expect the wrong sign. + +## Deliberately not solved + +- **The inversion itself.** Kept, for the reasons above, and now documented rather than silent. If it proves + worse in use than the magnitude signal is worth, the decision reverses to `K = 20` and the previous release's + gain is spent — that is the trade, stated so it can be made deliberately rather than discovered. +- **The thin pairs.** `architecture::component` at 8 rules and `seo::component` at 1 are a **rule-inventory + gap, not a scoring defect**. The coverage number is honest about how little is checked there. Closing it + means more rules in those pairs, which is the project's direction anyway. +- **Dilution.** 41 affected keys moving a category two points is the mean over 351 keys doing what a mean does. + Making the category reflect the _share_ of affected keys as well as their depth is a different aggregation + and a separate design. +- **Rendering `routes[].categories`.** Unchanged from the previous design; the HTML report and the dashboard + receive it and ignore it. + +## Testing + +This design changes documentation, so the tests are the docs-gating suites that already exist +(`packages/cli/test/docs-links.test.ts`, `rules-index.test.mjs`, `docs-embed.test.mjs`) plus one check the +prose cannot get from a test runner: + +**Does the prose answer the confusion that produced this design?** The field reader arrived at "severity has +stopped meaning anything" from a correct observation and an undocumented model. The documentation earns its +place only if a reader in that position comes away understanding why an `info` can cost more than a `warning` +and why that is not a defect. That is a field question, not a unit test, and this design should not be +implemented until it is answered. diff --git a/packages/core/src/reporter/json.ts b/packages/core/src/reporter/json.ts index 1768fcdea..ae8107851 100644 --- a/packages/core/src/reporter/json.ts +++ b/packages/core/src/reporter/json.ts @@ -1,5 +1,6 @@ import type { Category, Config, Result } from '../types.js'; -import { computeScore, computeHealth, scoresByCategory, type ScoreModel } from '../scoring/score.js'; +import { computeScore, computeHealth, scoresByCategory, INVENTORY_FLOOR, type ScoreModel } from '../scoring/score.js'; +import { buildInventory } from '../scoring/inventory.js'; import { summarize, effectiveSeverity, type Summary } from '../summary.js'; import { isPenalized } from '../rule.js'; @@ -33,11 +34,19 @@ export interface JsonReport { version: string; score: number; // combined Health score weights: Partial>; - categories: Record; + categories: Record; summary: Summary; rules: Record; routes: Array<{ route: string; score: number; categories: Record; issues: JsonIssue[] }>; siteIssues: JsonIssue[]; + /** + * Floored severity weight per `"::"` pair. Reproduces a `routes[].categories` entry + * (`100 - 100 * failed / inventories[pair]`): a key is either a route id or a source file path, and + * those two key spaces never overlap, so a category's results on one key always draw on a single scope. + * It does not reproduce `routes[].score`: a route spanning more than one pair sums their *raw* weights + * and floors that sum once, so adding this map's already-floored entries and re-dividing can disagree. + */ + inventories: Record; } function ruleEvidence( @@ -69,7 +78,10 @@ export function buildJsonReport( const rules = ruleEvidence(results, config, ruleIds); const categories = Object.fromEntries( - Object.entries(byCat).map(([cat, sr]) => [cat, { score: sr.score, scoreModel: sr.scoreModel }]) + Object.entries(byCat).map(([cat, sr]) => [ + cat, + { score: sr.score, scoreModel: sr.scoreModel, keys: sr.keys, affectedKeys: sr.affectedKeys } + ]) ); const routeMap = new Map(); @@ -98,7 +110,11 @@ export function buildJsonReport( .filter((r) => r.route === undefined && isPenalized(r.detection, config.treatDynamicAs)) .map((r) => ({ ...issueOf(r), severity: effectiveSeverity(r, config) })); - return { version: meta.version, score: health, weights, categories, summary, rules, routes, siteIssues }; + const inventories = Object.fromEntries( + [...buildInventory(config)].map(([pair, weight]) => [pair, Math.max(weight, INVENTORY_FLOOR)]) + ); + + return { version: meta.version, score: health, weights, categories, summary, rules, routes, siteIssues, inventories }; } /** Render results as the documented JSON report string (design §7). */ diff --git a/packages/core/src/scoring/score.ts b/packages/core/src/scoring/score.ts index 25c13a7c7..525241bbc 100644 --- a/packages/core/src/scoring/score.ts +++ b/packages/core/src/scoring/score.ts @@ -8,6 +8,15 @@ import { buildInventory, ruleScopes, DEDUCTION, type PairKey } from './inventory const CRITICAL_CAP = 79; +/** + * A key is never scored against less than this much severity weight. Without it a pair holding one rule + * makes that rule's finding cost the whole key, and a finding's cost stops tracking its severity — an + * `info` in an eight-rule pair outweighed a `warning` in a twenty-six-rule one. It also makes the + * denominator provably non-zero, which is why the division below needs no zero-guard of its own — + * a floor lowered below 1 would bring that guard back. + */ +export const INVENTORY_FLOOR = 25; + export interface ScoreModel { routeAverage: number; sitePenalty: number; @@ -25,6 +34,10 @@ export interface ScoreResult { */ rawScore: number; scoreModel: ScoreModel; + /** Keys this result set touched. */ + keys: number; + /** Keys carrying at least one penalized finding. */ + affectedKeys: number; } export interface ScoreOptions { @@ -70,11 +83,13 @@ export function computeScore(results: Result[], config: Config, options: ScoreOp } // Deficit space, as `computeHealth` already works: a mean of key scores computes - // 49.99999999999999 for a true 50 on two keys of deficit 300/9 and 600/9. + // 49.99999999999999 for a true 50 on two keys of deficit 1000/28 and 1800/28. + let affectedKeys = 0; let totalDeficit = 0; for (const [key, pairs] of observed) { let failed = 0; for (const d of ruleMax.get(key)?.values() ?? []) failed += d; + if (failed > 0) affectedKeys += 1; let inventoryWeight = 0; // `pairOf` and `inventory` are built from the same filtered `rules`, so every pair reachable // through `pairOf` also has an inventory entry; the `?? 0` is a degrade-to-0-not-NaN fallback @@ -82,11 +97,12 @@ export function computeScore(results: Result[], config: Config, options: ScoreOp for (const p of pairs) inventoryWeight += inventory.get(p) ?? 0; // `max` keeps `failed` from exceeding its own denominator, so a penalized finding can never still // score 100: the two ways that would happen are `treatDynamicAs: 'warn'` promoting a result's - // severity above its rule's, and a result whose rule is absent from the inventory. - inventoryWeight = Math.max(inventoryWeight, failed); + // severity above its rule's, and a result whose rule is absent from the inventory. The + // `INVENTORY_FLOOR` guard keeps a small pair from making one finding cost the whole key. + inventoryWeight = Math.max(inventoryWeight, failed, INVENTORY_FLOOR); // `100 - (100 * f) / i`, never `100 * (1 - f / i)`: the latter gives 19.999999999999996 for // f = 88, i = 110 and displays 19 for a true 20. - totalDeficit += inventoryWeight === 0 ? 0 : (100 * failed) / inventoryWeight; + totalDeficit += (100 * failed) / inventoryWeight; } const keyCount = observed.size; @@ -114,7 +130,13 @@ export function computeScore(results: Result[], config: Config, options: ScoreOp const criticalCap = capBinds ? CRITICAL_CAP : null; const rawScore = clamp(capBinds ? CRITICAL_CAP : rawUncapped); - return { score: Math.floor(rawScore), rawScore, scoreModel: { routeAverage, sitePenalty, criticalCap } }; + return { + score: Math.floor(rawScore), + rawScore, + scoreModel: { routeAverage, sitePenalty, criticalCap }, + keys: keyCount, + affectedKeys + }; } /** Compute an independent score per category present in `results` (issue #10). */ diff --git a/packages/core/test/html-report.test.ts b/packages/core/test/html-report.test.ts index ea70f2ab7..a72cb0b8a 100644 --- a/packages/core/test/html-report.test.ts +++ b/packages/core/test/html-report.test.ts @@ -19,11 +19,12 @@ const report: JsonReport = { score: 82, weights: { seo: 1, performance: 1 }, categories: { - seo: { score: 91, scoreModel: model() }, - performance: { score: 68, scoreModel: model() } + seo: { score: 91, scoreModel: model(), keys: 0, affectedKeys: 0 }, + performance: { score: 68, scoreModel: model(), keys: 0, affectedKeys: 0 } }, summary: { critical: 1, warning: 2, info: 1, passed: 37, dynamic: 3 }, rules: {}, + inventories: {}, routes: [ { route: '/', score: 100, categories: {}, issues: [] }, { @@ -260,7 +261,7 @@ describe('safety hardening (buildHtmlDocument is a public API; JsonReport is loo it('an attacker-controlled category key cannot appear as raw markup', () => { const evil: JsonReport = { ...report, - categories: { '': { score: 50, scoreModel: model() } }, + categories: { '': { score: 50, scoreModel: model(), keys: 0, affectedKeys: 0 } }, weights: {}, routes: [], siteIssues: [] diff --git a/packages/core/test/json-report.test.ts b/packages/core/test/json-report.test.ts index 46be9e267..c3d13ba19 100644 --- a/packages/core/test/json-report.test.ts +++ b/packages/core/test/json-report.test.ts @@ -1,5 +1,6 @@ import { describe, it, expect } from 'vitest'; -import { buildJsonReport, formatJsonReport, computeHealth, defineConfig, type Result } from '../src/index.js'; +import { buildJsonReport, formatJsonReport, computeHealth, defineConfig, allRules, type Result } from '../src/index.js'; +import { DEDUCTION } from '../src/scoring/inventory.js'; const config = defineConfig({}); const results: Result[] = [ @@ -235,3 +236,96 @@ describe('buildJsonReport — per-route category scores', () => { expect(report.routes[0]!.categories).toEqual({ seo: 95, performance: 100 }); }); }); + +describe('buildJsonReport — category reach', () => { + it('reports keys and affectedKeys per category', () => { + const results: Result[] = [ + { + id: 'seo/canonical-url', + category: 'seo', + severity: 'warning', + detection: { presence: 'none', value: 'absent' }, + route: '/a', + message: 'x' + }, + { + id: 'seo/canonical-url', + category: 'seo', + severity: 'warning', + detection: { presence: 'own', value: 'static' }, + route: '/b', + message: 'ok' + } + ]; + const report = buildJsonReport(results, config, { version: '0.0.0' }); + expect(report.categories.seo!.keys).toBe(2); + expect(report.categories.seo!.affectedKeys).toBe(1); + }); +}); + +describe('buildJsonReport — pair inventories', () => { + it('reports the floored weight of each pair', () => { + const results: Result[] = [ + { + id: 'seo/canonical-url', + category: 'seo', + severity: 'warning', + detection: { presence: 'none', value: 'absent' }, + route: '/a', + message: 'x' + } + ]; + const report = buildJsonReport(results, config, { version: '0.0.0' }); + // seo::route holds 110 points and is above the floor; architecture::component holds 8 and is not. + expect(report.inventories['seo::route']).toBe(110); + expect(report.inventories['architecture::component']).toBe(25); + }); + + it('lets a reader recompute a route category score from the map', () => { + const results: Result[] = [ + { + id: 'seo/canonical-url', + category: 'seo', + severity: 'warning', + detection: { presence: 'none', value: 'absent' }, + route: '/a', + message: 'x' + } + ]; + const report = buildJsonReport(results, config, { version: '0.0.0' }); + const i = report.inventories['seo::route']!; + expect(Math.floor(100 - (100 * 5) / i)).toBe(report.routes[0]!.categories.seo); + }); + + it('recomputes every present category from `inventories`, including one that spans both scopes', () => { + // The recomputation above holds only because a route id and a source file path are different key + // spaces, so a category's results on one key always draw on a single scope — never asserted directly + // against the registry until now. Built from `allRules` itself (not a hard-coded id list): `performance` + // and `seo` each carry both a route-scoped and a component-scoped rule for real, so this is the case + // the precondition has to hold for, not a synthetic one. + const nonProjectRules = allRules.filter((r) => r.scope !== 'project'); + const results: Result[] = nonProjectRules.map((r) => ({ + id: r.id, + category: r.category, + severity: r.severity, + detection: { presence: 'none', value: 'absent' }, + route: r.scope === 'route' ? '/route-key' : 'src/component-key.svelte', + message: 'x' + })); + const report = buildJsonReport(results, config, { version: '0.0.0' }); + + for (const route of report.routes) { + const scope = route.route === '/route-key' ? 'route' : 'component'; + for (const [category, score] of Object.entries(route.categories)) { + const failed = nonProjectRules + .filter((r) => r.category === category && r.scope === scope) + .reduce((sum, r) => sum + DEDUCTION[r.severity], 0); + const inventory = report.inventories[`${category}::${scope}`]!; + expect(Math.floor(100 - (100 * failed) / inventory)).toBe(score); + } + } + // Confirm the case above is actually exercised, not vacuously true because no category spans both keys. + expect(report.routes.find((r) => r.route === '/route-key')!.categories).toHaveProperty('performance'); + expect(report.routes.find((r) => r.route === 'src/component-key.svelte')!.categories).toHaveProperty('performance'); + }); +}); diff --git a/packages/core/test/score.test.ts b/packages/core/test/score.test.ts index 017a39986..6d5802481 100644 --- a/packages/core/test/score.test.ts +++ b/packages/core/test/score.test.ts @@ -2,6 +2,7 @@ import { describe, it, expect } from 'vitest'; import { computeScore, defineConfig, scoresByCategory, type Result } from '../src/index.js'; import type { Rule } from '../src/rule.js'; +import { buildInventory, DEDUCTION } from '../src/scoring/inventory.js'; const pass = (id: string, route: string): Result => ({ id, @@ -256,55 +257,30 @@ describe('computeScore — a displayed 100 means zero deduction', () => { }); it('decides the cap on the raw value, not the floored one', () => { - // 10 keys, sitePenalty 0. Keys 0-8: a critical + a warning on two distinct rule ids - // (deduction 15+5=20, route score 80). Key 9: a critical + a warning + an info on three - // distinct rule ids (deduction 15+5+1=21, route score 79). Distinct ids per key so the - // deductions sum instead of only the max of duplicates counting. - // Mean = (9*80 + 79)/10 = 79.9. - // Deciding on the raw value: 79.9 > 79 -> cap binds -> rawScore 79, criticalCap 79. - // Deciding on the floored value: floor(79.9) - 0 = 79 -> 79 > 79 is false -> no cap -> - // rawScore would stay 79.9 and criticalCap null. Displayed `score` is 79 either way (floor(79.9) - // is also 79), which is why this test must assert on rawScore/criticalCap, not on score. - const results: Result[] = Array.from({ length: 10 }, (_, i) => { - const route = `/k${i}`; - const findings: Result[] = [ - { - id: 'seo/title-presence', - category: 'seo' as const, - severity: 'critical' as const, - detection: { presence: 'none' as const, value: 'absent' as const }, - route, - message: 'm', - recommendation: 'r' - }, - { - id: 'seo/description-presence', - category: 'seo' as const, - severity: 'warning' as const, - detection: { presence: 'none' as const, value: 'absent' as const }, - route, - message: 'm', - recommendation: 'r' - } - ]; - if (i === 9) { - findings.push({ - id: 'seo/canonical-url', - category: 'seo' as const, - severity: 'info' as const, - detection: { presence: 'none' as const, value: 'absent' as const }, - route, - message: 'm', - recommendation: 'r' - }); - } - return findings; - }).flat(); + // One route, one pair (seo::route), inventory padded to exactly 200 with filler info rules + // that never appear in results — they exist only to size the denominator. A critical, five + // warnings and one info fail: failed = 15 + 5*5 + 1 = 41 of 200. + // Raw route average = 100 - 4100/200 = 79.5 (exact: 41/2 is a power-of-two fraction). + // Deciding on the raw value: 79.5 > 79 -> cap binds -> rawScore 79, criticalCap 79. + // Deciding on the floored value: floor(79.5) = 79 -> 79 > 79 is false -> no cap -> rawScore + // would stay 79.5 and criticalCap null. Displayed `score` is 79 either way (floor(79.5) is + // also 79), which is why this test must assert on rawScore/criticalCap, not on score. + const rules = [ + r('s/c1', 'seo', 'route', 'critical'), + ...Array.from({ length: 5 }, (_, i) => r(`s/w${i}`, 'seo', 'route', 'warning')), + r('s/i0', 'seo', 'route', 'info'), + ...Array.from({ length: 159 }, (_, i) => r(`s/fill${i}`, 'seo', 'route', 'info')) + ]; // inventory: 15 + 5*5 + 1 + 159 = 200 + const results: Result[] = [ + fail('s/c1', '/a', 'critical'), + ...Array.from({ length: 5 }, (_, i) => fail(`s/w${i}`, '/a', 'warning')), + fail('s/i0', '/a', 'info') + ]; - const r = computeScore(results, CONFIG); - expect(r.score).toBe(79); - expect(r.rawScore).toBe(79); - expect(r.scoreModel.criticalCap).toBe(79); + const result = computeScore(results, defineConfig({}), { rules }); + expect(result.score).toBe(79); + expect(result.rawScore).toBe(79); + expect(result.scoreModel.criticalCap).toBe(79); }); }); @@ -324,15 +300,17 @@ describe('computeScore — proportional model', () => { const config = defineConfig({}); it('scores a key as the share of its pair that is intact', () => { - // failedWeight 5 of inventory 9 -> 100 - 500/9 = 44.44…, floored once at the category. + // failedWeight 5 of inventory 9, floored to 25 -> 100 - 500/25 = 80. const results = [fail('p/w1', 'src/A.svelte', 'warning')]; const { score } = computeScore(results, config, { rules: PERF }); - expect(score).toBe(44); + expect(score).toBe(80); }); - it('lets a key reach 0 when everything in its pair fails', () => { + it('keeps a floored pair from reaching 0 when everything in it fails', () => { + // failedWeight 9 of inventory 9, floored to 25 -> 100 - 900/25 = 64: a fully failed small pair + // still costs less than the whole key. const results = PERF.map((rule) => fail(rule.id, 'src/A.svelte', rule.severity as 'warning' | 'info')); - expect(computeScore(results, config, { rules: PERF, applyCriticalCap: false }).score).toBe(0); + expect(computeScore(results, config, { rules: PERF, applyCriticalCap: false }).score).toBe(64); }); it('distinguishes one affected key from many', () => { @@ -347,11 +325,22 @@ describe('computeScore — proportional model', () => { }); it('sums the inventory over every pair observed on a key', () => { - // One seo route warning beside a passing performance route rule: 100 - 500/(5+5) = 50, - // where the seo pair alone would give 100 - 500/5 = 0. - const rules = [r('seo/x', 'seo', 'route', 'warning'), r('perf/y', 'performance', 'route', 'warning')]; + // Two 30-point pairs (each ≥ 25, so the floor is a no-op): a seo::route warning beside a + // passing performance::route rule sums to 60 -> 100 - 500/60 = 91, where the seo pair alone + // would give 100 - 500/30 = 83. A third, unreferenced architecture::route rule shares the same + // scope but a different category: it must not be pulled into the sum, so it also pins the + // pairing to (category, scope), not scope alone. + const seoRoute = [ + r('seo/x', 'seo', 'route', 'warning'), + ...Array.from({ length: 25 }, (_, i) => r(`seo/fill${i}`, 'seo', 'route', 'info')) + ]; // 5 + 25 = 30 + const perfRoute = [ + r('perf/y', 'performance', 'route', 'warning'), + ...Array.from({ length: 25 }, (_, i) => r(`perf/fill${i}`, 'performance', 'route', 'info')) + ]; // 5 + 25 = 30 + const rules = [...seoRoute, ...perfRoute, r('arch/unreferenced', 'architecture', 'route', 'warning')]; const results = [fail('seo/x', '/a', 'warning'), pass('perf/y', '/a')]; - expect(computeScore(results, defineConfig({}), { rules }).score).toBe(50); + expect(computeScore(results, defineConfig({}), { rules }).score).toBe(91); }); it('keeps an integral score integral', () => { @@ -394,35 +383,38 @@ describe('computeScore — proportional model', () => { expect(computeScore(results, defineConfig({}), { rules, applyCriticalCap: false }).score).toBe(44); }); - it('scores 50 for a two-key mean of failed weight 2 and 9 against inventory 11', () => { - // Two keys sharing an inventory of 11 (11 info rules), failing weight 2 and 9: true mean is - // exactly 50. Either wrong evaluation order lands the mean at 49.99999999999999, displaying 49. - const rules = Array.from({ length: 11 }, (_, i) => r(`p/i${i}`, 'performance', 'component', 'info')); + it('scores 50 for a two-key mean of failed weight 10 and 20 against an inventory of 30', () => { + // Inventory 30 is already ≥ 25, so this anchors the deficit-space mean itself, not the floor: + // deficits 1000/30 and 2000/30 average in deficit space to exactly 50. A mean of key scores + // computes 49.99999999999999 for the same true 50 (verified against the formula, not the runner). + const rules = Array.from({ length: 30 }, (_, i) => r(`p/i${i}`, 'performance', 'component', 'info')); const results = [ - ...Array.from({ length: 2 }, (_, i) => fail(`p/i${i}`, 'src/A.svelte', 'info')), - ...Array.from({ length: 9 }, (_, i) => fail(`p/i${i}`, 'src/B.svelte', 'info')) + ...Array.from({ length: 10 }, (_, i) => fail(`p/i${i}`, 'src/A.svelte', 'info')), + ...Array.from({ length: 20 }, (_, i) => fail(`p/i${i}`, 'src/B.svelte', 'info')) ]; expect(computeScore(results, defineConfig({}), { rules }).score).toBe(50); }); it('keeps an integral mean integral across keys', () => { - // Two keys, deficits 300/9 and 600/9, true mean exactly 50. A mean of key scores gives - // 49.99999999999999 and displays 49. + // Inventory 28 (28 info rules) is already ≥ 25, so this anchors the deficit-space mean and the + // evaluation order together, not the floor: two keys failing 10 and 18 of the infos (deficit + // 1000/28 and 1800/28) average in deficit space to exactly 50. Both a mean of key scores and the + // `100 * (1 - f/i)` evaluation order compute 49.99999999999999 for the same true 50 (verified + // against the formula, not the runner). + const rules = Array.from({ length: 28 }, (_, i) => r(`m/i${i}`, 'performance', 'component', 'info')); const results = [ - fail('p/i1', 'src/A.svelte', 'info'), - fail('p/i2', 'src/A.svelte', 'info'), - fail('p/i3', 'src/A.svelte', 'info'), - fail('p/w1', 'src/B.svelte', 'warning'), - fail('p/i1', 'src/B.svelte', 'info') + ...Array.from({ length: 10 }, (_, i) => fail(`m/i${i}`, 'src/A.svelte', 'info')), + ...Array.from({ length: 18 }, (_, i) => fail(`m/i${i + 10}`, 'src/B.svelte', 'info')) ]; - expect(computeScore(results, config, { rules: PERF }).score).toBe(50); + expect(computeScore(results, config, { rules }).score).toBe(50); }); - it('scores 0, not NaN, for a penalized result whose rule is not in the inventory', () => { + it('scores 80, not NaN, for a penalized result whose rule is not in the inventory', () => { + // No pair observed for `ghost/rule`, so the denominator floors straight to 25: 100 - 500/25. const results = [fail('ghost/rule', 'src/A.svelte', 'warning')]; const { score } = computeScore(results, config, { rules: PERF, applyCriticalCap: false }); expect(Number.isFinite(score)).toBe(true); - expect(score).toBe(0); + expect(score).toBe(80); }); it('scores 100 for a key whose only results come from rules outside the inventory', () => { @@ -430,23 +422,48 @@ describe('computeScore — proportional model', () => { expect(computeScore(results, config, { rules: PERF }).score).toBe(100); }); + it('lets failed weight outrun the floor when two ghost rules fail on the same key', () => { + // Two rule ids outside the inventory, both critical, on one key: failed = 15 + 15 = 30, above + // both the observed inventory (0, no pair) and the floor (25) — the case `max(inventoryWeight, + // failed, INVENTORY_FLOOR)` needs `failed` for, not just the floor. Diluted across 99 clean keys + // so the excess shows up in the mean instead of vanishing into the final clamp: with `failed` in + // the max, this key's deficit is 100, giving 100 - 100/100 = 99; dropping `failed` from the max + // would floor that key's denominator to 25 instead of 30, giving deficit 120 and 100 - 120/100 = 98. + const cleanKeys = Array.from({ length: 99 }, (_, i) => `src/${i}.svelte`); + const results: Result[] = [ + fail('ghost/one', 'src/bad.svelte', 'critical'), + fail('ghost/two', 'src/bad.svelte', 'critical'), + ...cleanKeys.map((k) => pass('p/i1', k)) + ]; + expect(computeScore(results, config, { rules: PERF, applyCriticalCap: false }).score).toBe(99); + }); + it('narrowing the rule set to one category leaves that category unchanged', () => { - const mixed = [...PERF, r('seo/x', 'seo', 'route', 'warning')]; - const results = [fail('p/w1', 'src/A.svelte', 'warning'), pass('p/i1', 'src/A.svelte')]; + // PERF's own inventory (9) floors to 25 regardless of what else is in `rules`, so a mutant that + // pools every pair into one denominator would still pass here. Use a performance::component + // inventory of 30 instead — already above the floor, so pooling in an unrelated seo::route + // rule's weight (5) would move the score (83 -> 85) if pools were not partitioned by pair. + const perf30 = [ + r('p/w1', 'performance', 'component', 'warning'), + ...Array.from({ length: 25 }, (_, i) => r(`p/fill${i}`, 'performance', 'component', 'info')) + ]; // 5 + 25 = 30 + const mixed = [...perf30, r('seo/x', 'seo', 'route', 'warning')]; + const results = [fail('p/w1', 'src/A.svelte', 'warning')]; expect(computeScore(results, config, { rules: mixed }).score).toBe( - computeScore(results, config, { rules: PERF }).score + computeScore(results, config, { rules: perf30 }).score ); }); it('keeps a category with two scopes from merging them', () => { // A component key must not be measured against route-scoped rules it can never trigger. - // Merged, the inventory would be 5 + 45 and the key would score 90 instead of 0. + // Its own pair (5) floors to 25 -> score 80. Merged, the inventory would be 5 + 45 = 50 + // (above the floor) and the key would score 90 instead. const rules = [ r('p/comp', 'performance', 'component', 'warning'), ...Array.from({ length: 9 }, (_, i) => r(`p/route${i}`, 'performance', 'route', 'warning')) ]; const results = [fail('p/comp', 'src/A.svelte', 'warning')]; - expect(computeScore(results, config, { rules, applyCriticalCap: false }).score).toBe(0); + expect(computeScore(results, config, { rules, applyCriticalCap: false }).score).toBe(80); }); it('scores 100 when nothing is penalized', () => { @@ -471,40 +488,193 @@ describe('computeScore — proportional model', () => { describe('computeScore — inventory wiring to config (no rules option, real registry)', () => { it('a config severity override changes the default inventory, not just buildInventory in isolation', () => { - // architecture::component is 8 info rules (inventory 8) by default. Raising one rule's severity - // to 'warning' makes it 7 info + 1 warning = 12. Verified against the real registry: a key - // failing that one rule (now 'warning', deduction 5) scores 100 - 500/12 = 58.33, floored 58. - const config = defineConfig({ rules: { 'architecture/component-size': 'warning' } }); - const results = [fail('architecture/component-size', 'src/A.svelte', 'warning')]; + // architecture::component is 8 info rules (inventory 8) by default. Raising two of them to + // 'critical' makes it 6 info + 2 critical = 6 + 30 = 36, above the floor. Verified against the + // real registry: a key failing one of the raised rules (now 'critical', deduction 15) scores + // 100 - 1500/36 = 58.33, floored 58. A `computeScore` that ignored `config.rules` when building + // the inventory would see the unconfigured 8, floor it to 25, and score 100 - 1500/25 = 40 instead. + const config = defineConfig({ + rules: { 'architecture/component-size': 'critical', 'architecture/prop-count': 'critical' } + }); + const results = [fail('architecture/component-size', 'src/A.svelte', 'critical')]; expect(computeScore(results, config).score).toBe(58); }); }); describe('computeScore — a disabled rule injected via options.rules', () => { - it('scores 0 and finite for a rule turned off in config but still carried in options.rules', () => { + it('scores 80 and finite for a rule turned off in config but still carried in options.rules', () => { // `computeScore` now runs `options.rules` through `selectRules` itself, so an `off` rule never - // reaches the inventory or `pairOf`: its penalized result observes no pair, `Math.max` floors the - // denominator to `failed`, and the key scores 0 rather than NaN or a false 100. + // reaches the inventory or `pairOf`: its penalized result observes no pair, so the denominator + // floors to 25 (not `failed`), and the key scores 80 (100 - 500/25) rather than NaN or a false 100. const rule = r('a/off', 'architecture', 'component', 'warning'); const config = defineConfig({ rules: { 'a/off': 'off' } }); const results = [fail('a/off', 'src/A.svelte', 'warning')]; const { score } = computeScore(results, config, { rules: [rule] }); expect(Number.isFinite(score)).toBe(true); - expect(score).toBe(0); + expect(score).toBe(80); }); - it('scores 0, not 80, when the off rule shares its pair with an enabled rule', () => { + it('scores 96, matching the isolated case, when the off rule shares its pair with an enabled rule', () => { // Before `options.rules` was filtered, `buildInventory` dropped the off rule internally but // `ruleScopes` did not: charged against the enabled rule's own inventory of 5, the same failing // off rule scored 80 here and 0 when injected alone in its pair — an inconsistency for identical - // input. Filtering both from the same list makes the off rule invisible to both, so this must - // match the isolated case. + // input. Filtering both from the same list makes the off rule invisible to both: no pair is + // observed either way, so the denominator floors to 25 and the info deduction (1) gives + // 100 - 100/25 = 96 in both the isolated and the shared-pair case. const off = r('p/off', 'performance', 'component', 'info'); const enabled = r('p/warn', 'performance', 'component', 'warning'); const config = defineConfig({ rules: { 'p/off': 'off' } }); const results = [fail('p/off', 'src/A.svelte', 'info')]; const { score } = computeScore(results, config, { rules: [off, enabled], applyCriticalCap: false }); expect(Number.isFinite(score)).toBe(true); - expect(score).toBe(0); + expect(score).toBe(96); + }); +}); + +describe('computeScore — inventory floor', () => { + const config = defineConfig({}); + + it('scores a one-rule pair at 80 for a single warning', () => { + // Inventory 5 floored to 25: 100 − 500/25. + const rules = [r('x/only', 'seo', 'component', 'warning')]; + const results = [fail('x/only', 'src/A.svelte', 'warning')]; + expect(computeScore(results, config, { rules, applyCriticalCap: false }).score).toBe(80); + }); + + it('scores an eight-info pair at 96 for a single info', () => { + // Inventory 8 floored to 25: 100 − 100/25. + const rules = Array.from({ length: 8 }, (_, i) => r(`a/${i}`, 'architecture', 'component', 'info')); + const results = [fail('a/0', 'src/A.svelte', 'info')]; + expect(computeScore(results, config, { rules }).score).toBe(96); + }); + + it('leaves a pair above the floor unchanged', () => { + // Inventory 30 > 25, so one warning costs 500/30 and the key scores 83. + const rules = Array.from({ length: 6 }, (_, i) => r(`s/${i}`, 'seo', 'route', 'warning')); + const results = [fail('s/0', '/a', 'warning')]; + expect(computeScore(results, config, { rules }).score).toBe(83); + }); + + it('still scores 100 when nothing is penalized', () => { + const rules = [r('x/only', 'seo', 'component', 'warning')]; + expect(computeScore([pass('x/only', 'src/A.svelte')], config, { rules }).score).toBe(100); + }); + + it('still subtracts sitePenalty in absolute points', () => { + // A site-wide warning costs 5, not a share of anything — the floor must not reach it. + const rules = [r('p/site', 'seo', 'project', 'warning'), r('s/one', 'seo', 'route', 'warning')]; + const results = [ + { + id: 'p/site', + category: 'seo', + severity: 'warning', + detection: { presence: 'none', value: 'absent' }, + message: 'x' + }, + pass('s/one', '/a') + ] as Result[]; + const sr = computeScore(results, config, { rules }); + expect(sr.scoreModel.sitePenalty).toBe(5); + expect(sr.score).toBe(95); + }); + + it('orders info below warning in every registry pair', () => { + // Derived from the registry rather than written out, so a new rule cannot silently break the + // ordering. Each pair's real weight is rebuilt as a same-weight synthetic rule set and scored + // through computeScore itself — reimplementing the cost formula here would only rubber-stamp + // the same bug it is meant to catch. + const inv = buildInventory(config); + const infoScores: number[] = []; + const warnScores: number[] = []; + for (const [pair, weight] of inv) { + const [category, scope] = pair.split('::') as [Rule['category'], Rule['scope']]; + + const infoId = `${pair}/i0`; + const infoRule = r(infoId, category, scope, 'info'); + const infoRules = [ + infoRule, + ...Array.from({ length: weight - 1 }, (_, i) => r(`${pair}/i${i + 1}`, category, scope, 'info')) + ]; + infoScores.push( + computeScore([fail(infoId, 'k', 'info')], config, { rules: infoRules, applyCriticalCap: false }).score + ); + + // Every rule not designated the failing warning is an info filler, so the set still totals `weight`. + const warnId = `${pair}/w0`; + const warnRule = r(warnId, category, scope, 'warning'); + const remainder = weight - 5; + const warnRules = [ + warnRule, + ...Array.from({ length: remainder }, (_, i) => r(`${pair}/r${i}`, category, scope, 'info')) + ]; + warnScores.push( + computeScore([fail(warnId, 'k', 'warning')], config, { rules: warnRules, applyCriticalCap: false }).score + ); + } + // The worst (lowest-scoring) info finding anywhere in the registry must still score no worse + // than the best (highest-scoring) warning finding anywhere in it. + expect(Math.min(...infoScores)).toBeGreaterThanOrEqual(Math.max(...warnScores)); + }); + + it('orders severities within a pair, but not across pairs', () => { + // Extends the previous test's construction to `critical` and to comparing across pairs, not + // only within one — pinning the actual guarantee (per pair) against the stronger one it is + // easy to misstate (per category). + const inv = buildInventory(config); + const scoreOf = (pair: string, weight: number, sev: 'critical' | 'warning' | 'info') => { + const [category, scope] = pair.split('::') as [Rule['category'], Rule['scope']]; + const markerId = `${pair}/${sev}`; + const filler = Math.max(weight - DEDUCTION[sev], 0); + const rules = [ + r(markerId, category, scope, sev), + ...Array.from({ length: filler }, (_, i) => r(`${pair}/f${i}`, category, scope, 'info')) + ]; + return computeScore([fail(markerId, 'k', sev)], config, { rules, applyCriticalCap: false }).score; + }; + + const warningScores: number[] = []; + const criticalScores: number[] = []; + for (const [pair, weight] of inv) { + const info = scoreOf(pair, weight, 'info'); + const warning = scoreOf(pair, weight, 'warning'); + const critical = scoreOf(pair, weight, 'critical'); + expect(info).toBeGreaterThan(warning); + expect(warning).toBeGreaterThan(critical); + warningScores.push(warning); + criticalScores.push(critical); + } + // A warning in the thinnest (floored) pair still scores lower — costs more — than a critical + // in the thickest one: the order above holds inside a pair, and is not guaranteed across pairs. + expect(Math.min(...warningScores)).toBeLessThan(Math.max(...criticalScores)); + }); +}); + +describe('computeScore — reach', () => { + const config = defineConfig({}); + const rules = Array.from({ length: 8 }, (_, i) => r(`a/${i}`, 'architecture', 'component', 'info')); + + it('counts keys touched and keys penalized', () => { + const results = [fail('a/0', 'src/A.svelte', 'info'), pass('a/0', 'src/B.svelte'), pass('a/0', 'src/C.svelte')]; + const sr = computeScore(results, config, { rules }); + expect(sr.keys).toBe(3); + expect(sr.affectedKeys).toBe(1); + }); + + it('reports the same score and different reach for one finding and for many', () => { + // The reason reach exists: 41 affected keys of 351 (a real project's own numbers) dilute to + // the same floored score as a single affected key, and must still be distinguishable. + const keys = Array.from({ length: 351 }, (_, i) => `src/${i}.svelte`); + const one = keys.map((k, i) => (i === 0 ? fail('a/0', k, 'info') : pass('a/0', k))); + const many = keys.map((k, i) => (i < 41 ? fail('a/0', k, 'info') : pass('a/0', k))); + const a = computeScore(one, config, { rules }); + const b = computeScore(many, config, { rules }); + expect(a.score).toBe(b.score); + expect(a.affectedKeys).toBe(1); + expect(b.affectedKeys).toBe(41); + }); + + it('counts a key once however many rules penalize it', () => { + const results = [fail('a/0', 'src/A.svelte', 'info'), fail('a/1', 'src/A.svelte', 'info')]; + expect(computeScore(results, config, { rules }).affectedKeys).toBe(1); }); }); diff --git a/packages/vite/test/app-shell-static.test.ts b/packages/vite/test/app-shell-static.test.ts index 649b545a1..41096640a 100644 --- a/packages/vite/test/app-shell-static.test.ts +++ b/packages/vite/test/app-shell-static.test.ts @@ -14,9 +14,12 @@ const report: JsonReport = { version: '1', score: 50, weights: { seo: 1 }, - categories: { seo: { score: 50, scoreModel: { routeAverage: 50, sitePenalty: 0, criticalCap: null } } }, + categories: { + seo: { score: 50, scoreModel: { routeAverage: 50, sitePenalty: 0, criticalCap: null }, keys: 0, affectedKeys: 0 } + }, summary: { critical: 1, warning: 0, info: 0, passed: 0, dynamic: 0 } as never, rules: {}, + inventories: {}, routes: [ { route: '/blog/hello', diff --git a/packages/vite/test/ui-dashboard.test.ts b/packages/vite/test/ui-dashboard.test.ts index 1d74ff368..6883fc003 100644 --- a/packages/vite/test/ui-dashboard.test.ts +++ b/packages/vite/test/ui-dashboard.test.ts @@ -7,9 +7,10 @@ const baseSnapshot: DashboardSnapshot = { version: '1', score: 80, weights: { seo: 1 }, - categories: { seo: { score: 80, scoreModel: 'weighted' as never } }, + categories: { seo: { score: 80, scoreModel: 'weighted' as never, keys: 0, affectedKeys: 0 } }, summary: { critical: 0, warning: 0, info: 0, passed: 0, dynamic: 0 } as never, rules: {}, + inventories: {}, routes: [ { route: '/a',