From 73a658550e59b8d0ea81e7f5fd45d26acafd6946 Mon Sep 17 00:00:00 2001 From: Paddy Mullen Date: Sun, 20 Sep 2026 10:29:50 -0400 Subject: [PATCH 001/111] =?UTF-8?q?docs(plans):=20ADR-007,=20ADR-008,=20AD?= =?UTF-8?q?R-009=20=E2=80=94=20the=20cache=20redesign,=20as=20one=20set?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Brings the three draft ADRs from the 2026-09-18 cache audit, and the spikes they cite, onto one branch so they can be reviewed together. They were drafted as separate PRs (#180, #181, #182) and revised there during the 2026-09-19/20 design session; this commit is their state at the end of that session, unchanged. - ADR-007: tallyman owns result materialization (no xorq cache nodes in builds); Buckaroo is a displayer; every diff is built as an entry. - ADR-008: every file carries a visible __row_order column and every page request sorts by it. - ADR-009: materialization runs single-partition, and result_digest is a digest of the snapshot's content. Co-Authored-By: Claude Fable 5.1 --- .../ADR-007-tallyman-owned-materialization.md | 498 ++++++++++++++++++ plans/ADR-008-row-order-of-reads.md | 382 ++++++++++++++ plans/ADR-009-digest-stability.md | 268 ++++++++++ scripts/spike_bare_read_chaining.py | 117 ++++ scripts/spike_float_aggregate_digest.py | 77 +++ scripts/spike_logical_digest.py | 229 ++++++++ scripts/spike_row_order_paging.py | 159 ++++++ scripts/spike_window_read_order.py | 148 ++++++ 8 files changed, 1878 insertions(+) create mode 100644 plans/ADR-007-tallyman-owned-materialization.md create mode 100644 plans/ADR-008-row-order-of-reads.md create mode 100644 plans/ADR-009-digest-stability.md create mode 100644 scripts/spike_bare_read_chaining.py create mode 100644 scripts/spike_float_aggregate_digest.py create mode 100644 scripts/spike_logical_digest.py create mode 100644 scripts/spike_row_order_paging.py create mode 100644 scripts/spike_window_read_order.py diff --git a/plans/ADR-007-tallyman-owned-materialization.md b/plans/ADR-007-tallyman-owned-materialization.md new file mode 100644 index 0000000..404cd5d --- /dev/null +++ b/plans/ADR-007-tallyman-owned-materialization.md @@ -0,0 +1,498 @@ +# ADR: Tallyman owns result materialization (no xorq cache nodes in builds) + +- **Status:** Proposed (2026-09-18, revised 2026-09-20 in the grilling + session, which added the governing rule, decision D10, and the resolution + recorded under D5). Supersedes two decisions of + `plans/ADR-006-read-path-loads-builds.md`: its D4 (chaining inlines the + parent's cache node) and its D8 (the manifest records the snapshot key and + reads assert it). Five other ADR-006 decisions keep their intent: D5 (the + canonical sort), D6 (a missing build is a hard error), D7 (verification runs + in production and is loud), D10 (an unfaithful heal wipes the entry's + Buckaroo state) and D12 (unfaithful entries are pinned and badged). The last + three attach to `ensure_materialized`, which is this ADR's D5. +- **Reading decision labels:** a bare label such as "D5" in this document + always means this ADR's own decision. Another ADR's decision is always + written with its ADR number and a few words saying what it decides. +- **Context:** the 2026-09-18 cache audit (tallyman @ `a748ea6`, buckaroo + 0.15.4, xorq 0.3.26). One finding is filed upstream of tallyman: + buckaroo-data/buckaroo#972 (`/load_expr` has no `cache_dir`). The direction + was set by Paddy the same day: "I want to depend on xorq as little as + possible for caching." +- **Affected code:** `src/tallyman_xorq/source_cache.py` (`rewrite_for_build`), + `src/tallyman_xorq/result_cache.py` (`_resolve_result_plan`, + `cached_result_expr`, `entry_graph_expr`, `baked_snapshot_path`, + `_cached_node_path`, `_assert_recorded_snapshot_key`, `_verify_self_heal`), + `src/tallyman_xorq/build.py` (the execute-once step, `build.py:496-557`), + `src/tallyman_xorq/io.py` (`tracked_expr_from_alias`, + `pinned_expr_from_alias`), `src/tallyman_xorq/portable.py` + (`rewrite_cache_dirs`), `src/tallyman_core/manifest.py` (`snapshot_key`), + `src/tallyman_companion/buckaroo_lifecycle.py` (`load_session`), + `docs/system-contract.md`. +- **Related ADRs:** `plans/ADR-002-source-identity-content-hash.md` (content in + the path; D3 reuses the device), `plans/ADR-003-result-cache-cost-rubric.md` + (its motivating misclassification; see Consequences), + `plans/ADR-008-row-order-of-reads.md` and + `plans/ADR-009-digest-stability.md` (the other two hash- or digest-changing + decisions that share this ADR's corpus rebuild). +- **Evidence:** `scripts/spike_bare_read_chaining.py` (results under D3), and + the audit measurements quoted in the Problem section. + +## Terms + +- **Entry:** one catalog computation, stored under its content hash. +- **Build:** the `xorq_build/` directory inside an entry, which is the entry's + computation graph written to disk by xorq. +- **Replay a build:** load that directory back into a live expression and + execute it. +- **Materialize:** run an entry's computation once and write the result to a + parquet file. That file is the entry's **snapshot**, and it lives under + `compute_cache/result_cache/`. +- **Worthy entry:** an entry tallyman materializes, because its graph does work + that is expensive or that cannot inherit a row order (an aggregate, join, + sort, window function, UDF, union). **Cheap entry:** one it does not, whose + small plan re-runs on every read. +- **Cache node:** xorq's `CachedNode`, a marker inside a graph meaning "look for + a file with this key, and if it is missing run the graph below me and write + it". +- **Chaining:** a recipe building on another entry through + `tracked_expr_from_alias` or `pinned_expr_from_alias`. +- **Heal:** re-create a snapshot that is missing from disk by re-running the + entry's build. +- **Session:** one grid's state inside the Buckaroo server process. +- **View build:** a build whose whole graph is one step, "read this parquet + file". +- **Live diff:** the compare grid shown when two versions are diffed. A + **promoted diff** is one that has been saved as a catalog entry. +- **Checkpoint:** tallyman's step that zips new entries and makes one git + commit in the catalog repository. + +## Problem + +Tallyman's result cache is xorq's cache. `rewrite_for_build` +(`source_cache.py:241-243`) wraps every worthy expression in +`.cache(cache=ParquetSnapshotCache(...))`, and that one call decides everything +else: the file's name is xorq's snapshot key, its directory is a base path that +xorq does not serialize, its writer is xorq's `ParquetStorage`, it is written as +a side effect of `loaded.count().execute()` (`build.py:540`, +`result_cache.py:733`), a missing ancestor is regenerated by a nested cache node +noticing its own file is gone, and Buckaroo is handed a build that contains all +of this. + +The audit found the following, each a consequence of that arrangement. + +1. **Snapshots escape the project.** xorq serializes a cache node's + `relative_path` and never its base path, and `load_expr(cache_dir=...)` + redirects only the root cache node. + - Buckaroo's `/load_expr` calls `load_expr(build_dir)` with no `cache_dir`, + so the viewer never reads tallyman's snapshot. The first view re-executes + the whole graph inside Buckaroo and writes a second copy under + `~/.cache/xorq/result_cache/`. On the dev machine that directory held 68 + files and 14 GB, with the same keys and sizes as the project's + `compute_cache` (buckaroo#972). + - `build.py:496` loads the new build with `cache_dir` but without + `rewrite_cache_dirs`, so nested ancestor nodes resolve to `~/.cache/xorq` + while a child builds. Every child build re-executes its expensive parent + and leaves a duplicate there. Reproduced in scratch homes where Buckaroo + never ran. +2. **The writer is unsafe under concurrency.** `ParquetStorage` writes through + a fixed `.parquet.tmp`. Four concurrent cold writers of one key gave + three `FileNotFoundError`s and a corrupt final file, and because a hit is + decided by file existence the corrupt file is then served forever. + `_heal_lock` covers threads in one process only, builds take no lock, and + FastMCP runs sync tools in a threadpool, so parallel tool calls do build + concurrently. +3. **The file shape is poor.** One row group per DataFusion batch (at most + 8,192 rows), Snappy. A real 3.68 GB snapshot has 9,525 row groups and a + 44.9 MB footer. `cached_result_expr` constructs a fresh read on every call + (`result_cache.py:742`), so that footer is opened again for every page + request, and every call registers another table in the shared backend + (40 reads, 40 tables). +4. **The bytes depend on how the node was executed.** xorq stamps provenance + metadata only when the cache node is the root of the executed expression: + 1,234 bytes against 858 for the same rows. Build and heal agree today only + because both happen to call `count()`. +5. **Inlined chaining misclassifies children.** Under ADR-006 decision D4 + (chaining inlines the parent's cache node) a child's + graph contains its parent's Aggregate and canonical Sort, so a filter over a + cached aggregate is itself worthy and writes a full copy. This is ADR-003's + motivating bug. Graphs also grow with chain depth, which is the cost the + pre-#74 per-entry parquet boundary existed to avoid (#82). +6. **Tallyman carries code whose only job is to aim xorq's cache.** + `rewrite_cache_dirs`, the contract's "every loader must supply `cache_dir`", + `manifest.snapshot_key` with `_assert_recorded_snapshot_key` (ADR-006 + decision D8, the snapshot-key tripwire), + and `_cached_node_path`. + +xorq is frozen upstream for this project, so every item above needs a +tallyman-side workaround for as long as xorq's cache is in the path. + +## Governing rule + +Stated by Paddy in the grilling session (2026-09-19 and 2026-09-20), and the +reason several decisions below are consequences and not separate choices: + +> Buckaroo is a very good displayer. It runs queries only for summary stats, +> sorting and paging. Anything more is done by tallyman first. If something +> tallyman asked Buckaroo to display does not exist, that is on tallyman. + +> When the MCP asks tallyman to create an entry, tallyman should run the query +> and materialize the parquet if necessary immediately. Tallyman shouldn't call +> Buckaroo to display an entry until the original query has finished. + +> I want a cohesive system that works reliably, then we can worry about speed +> problems as they come up. We don't have a cohesive system now. + +That last statement sets the priority for all three ADRs of this set (this one, +`plans/ADR-008-row-order-of-reads.md` and `plans/ADR-009-digest-stability.md`): +where a uniform rule and a faster special case compete, the uniform rule is the +decision and the faster path is noted for later. + +So tallyman runs an entry's computation, or a diff, to completion before it +asks Buckaroo to show anything. No other process runs an entry's expensive +computation, writes result files, or repairs tallyman's cache. Besides being +simpler, this puts every failure of a computation in tallyman's process, where +it can be logged and reported, and none inside a grid query in Buckaroo. + +## Decisions + +### D1. Builds carry no cache nodes + +`rewrite_for_build` keeps the in-memory-read rejection and the canonical sort +(ADR-006 decision D5, so the sort stays inside the build) and stops calling +`.cache()`: + +```python +if _is_worthy_expr(expr): + expr = _canonical_sorted(expr) +``` + +The source-read injection (`source_cache.py:215-234`) is deleted with it. It has +no producer today: `deferred_read_csv` is a build error and no JSON reader +exists. A recipe that arrives already containing a `CachedNode` becomes a build +error next to the in-memory check, because a node with default storage would +write under `~/.cache/xorq`. + +`source_cache.py:229` and `:243` are the only places in `src/` that create a +cache node, so this one change removes xorq's cache from tallyman's data path. +xorq remains the expression, build, load, hashing and execution layer. + +Removing the node alone would turn every worthy entry into a recompute entry +(`_resolve_result_plan` treats "no `CachedNode` on top" as recompute, +`result_cache.py:642-649`), so D2 to D6 land in the same change. + +*Rejected:* patch ADR-006 decision D4 (inlined chaining) in place. Deep-rewrite +the build's load (a one-line +change, verified to stop the build leak with no hash change), wait for +buckaroo#972, put a cross-process lock around xorq's writer, and pre-seed +xorq's files with a better writer (a hit is a bare existence check, so tallyman +can write `.parquet` itself). This keeps every hash and needs no rebuild. +It also keeps the file's name as xorq's tokenization of the graph (so the +snapshot-key tripwire of ADR-006 decision D8 stays), keeps item 5, and leaves each fix as a workaround for +behaviour in a dependency that cannot be changed from here. + +### D2. A snapshot's location is a function of the content hash + +`snapshot_path(project, content_hash)` returns +`compute_cache/result_cache/.parquet`. Whether an entry has a +snapshot comes from the manifest's `cache_worthy`, as it does now. + +`cached_result_expr` keeps its name and signature (ADR-006 decision D2, "keep +the facade"). For a worthy +entry it returns one bare read of `snapshot_path`, memoized per +`(project, content_hash)` for the life of the process, so repeated reads stop +registering tables and stop re-opening the footer. A worthy entry whose file +exists is served without loading its build at all. For a cheap entry it returns +the loaded graph, as now. + +Deleted: `manifest.snapshot_key`, `_assert_recorded_snapshot_key`, +`_cached_node_path`, `rewrite_cache_dirs`, and the `cache_dir` argument. With +the path computed from the hash there is no second derivation for a tripwire to +compare against. + +*Rejected:* `-.parquet`, which would make a child name +its parent's exact bytes. An unfaithful heal (a heal whose result does not +match the recorded digest, ADR-006 decision D12) would then write a +file no existing child can find, and a flagged condition would become a hard +failure for the whole subtree. The digest stays in the manifest, where verify +already looks. + +### D3. Chaining through a worthy parent is a bare read of its snapshot + +`tracked_expr_from_alias` and `pinned_expr_from_alias` return +`deferred_read_parquet(snapshot_path(parent))` for a worthy parent, after +making sure the file exists (D5). For a cheap parent they return the parent's +loaded graph, which by D1 contains no cache nodes. `entry_graph_expr` and +`cached_result_expr` become the same function. + +The child's hash covers the literal path of the parent's snapshot, and that +path contains the parent's content hash. A child's identity is therefore a +function of its parent's identity, which is the device ADR-002 already uses for +sources: if you want content identity, put it in the path. + +Measured with `scripts/spike_bare_read_chaining.py` (raw xorq, no cache nodes): + +| Question | Result | +| --- | --- | +| Child hash, snapshot rewritten with other bytes at the same path | unchanged | +| Child hash, same bytes at another path | changes | +| `build_expr` of a child while the parent's snapshot is absent | raises `FileNotFoundError: local path does not exist` | +| `load_expr` of the child's build while the snapshot is absent | succeeds | +| Executing it in that state | raises `ValueError: At least one path is required` | +| Parent re-materialized from its own build, then the child executed | digest unchanged; child rows match the reference (19,777) | +| `classify_build` of a filter + computed column over the snapshot | cheap | +| `classify_build` of the same child with the parent graph inlined | worthy (`ops:Aggregate,Sort,SortKey`) | +| Child `expr.yaml` | carries the literal path; 3,434 bytes against 6,903 inlined | +| Files written under `XORQ_CACHE_DIR` | none | + +So the graph is cut at every worthy entry, a child of an aggregate is cheap, +and the literal path in `expr.yaml` means `make_portable_inplace` handles it +with the existing `${TALLYMAN_PROJECT_ROOT}` placeholder. + +*Rejected:* keep inlining the parent's graph without its cache node. Every read +of every descendant would re-run the parent's expensive subgraph. + +### D4. One writer, used by the build and by every heal + +`materialize(project, content_hash)` loads the entry's build, executes it as a +record-batch stream, and writes the snapshot itself: + +- a unique temp name in the destination directory, then `os.replace`; +- under a per-hash cross-process `flock`, with the existence check repeated + inside the lock, so a second process waits and then finds the file; +- it numbers the rows as it writes them, in a last column named `__row_order` + (decision D2 of `plans/ADR-008-row-order-of-reads.md`); +- it returns the digest of what it wrote. + +The build's execute-once step and every heal call this function, so the +contract's I2 ("result bytes are manufactured exactly once; a self-heal is +reproduce-and-verify") holds because there is one routine that manufactures +bytes. The file format and the digest definition are ADR-009's. + +*Rejected:* keep `ParquetStorage` behind a tallyman lock. That fixes the race +and keeps items 3 and 4. + +### D5. One entry point makes files exist: `ensure_materialized` + +`ensure_materialized(project, content_hash)` guarantees that every snapshot an +entry's plan reads is on disk before anything executes: + +1. If the entry is worthy and its snapshot exists, return. No build is loaded. +2. Otherwise load the entry's build and collect the snapshot paths its `Read` + nodes point at (any read under `compute_cache/result_cache/`). The list is + kept with the loaded plan in the existing LRU. +3. For each of those that is missing, recurse on the hash in its file name. +4. If the entry is worthy, `materialize` it. + +Every file it writes is verified against the manifest's `result_digest` before +it is served. A mismatch takes the existing path: a durable `unfaithful_heal` +record and the SSE event (ADR-006 decision D7, loud verification), the +stat-cache wipe and session eviction (ADR-006 decision D10), and the pin and +badge (ADR-006 decision D12). + +Callers: the canonical read (`cached_result_expr`, on every call, where step 1 +or a handful of `stat` calls is the whole cost, and which covers diff +composition), chaining at mint time (D3), `load_session` (D6), and the verify +sweep. + +ADR-006 decision D4 rejected bare-read chaining because "builds stay +non-self-contained and the pre-heal choreography stays load-bearing forever". +What that decision bought was a build that repairs its own ancestors when +something other than tallyman executes it cold. Buckaroo is the only such +thing, and the comment at `buckaroo_lifecycle.py:570-575` says the design +relied on it: "Buckaroo's replay of the build regenerates any evicted snapshot +through ordinary cache mechanics on first query." Under the governing rule +Buckaroo never does that, so the property has no user and nothing is given up. +This was question 1 of the grilling session, resolved 2026-09-20. + +The old pre-heal was also weaker than this function. It was a +`cached_result_expr` call with a discarded result at one call site, and +ancestors were healed only because reads then re-executed recipes. Here the +requirement is a precondition of the one canonical read that every in-process +consumer already uses (contract invariant I3, "one read semantics"), and it is +computed from the build's own reads, so it cannot drift from what execution +opens. + +*Rejected:* derive the required snapshots from `manifest.parents`. It needs only +JSON reads, but it depends on the recorded edges being complete and on each +parent's recorded worthiness matching what chaining did when the child was +minted. Those edges are recalc policy and are not complete today: a promoted +diff's recipe calls `build_diff_expr`, which reads both sides through +`cached_result_expr` and records no parent edge at all. The build's reads are +exactly what the engine will open. + +### D6. Buckaroo is handed something that already exists + +This is the governing rule applied to entry grids. + +For a worthy entry `load_session` calls `ensure_materialized`, then posts a +view build of the snapshot, written once to a stable per-entry directory +(Buckaroo's stat-cache keys include the build directory's path, which is why +the expanded build already lives at a stable path). Buckaroo never executes an +aggregate, join or sort on tallyman's behalf and never writes a snapshot, and +tallyman stops depending on buckaroo#972, whose premise (teach Buckaroo where +tallyman's cache is) the rule contradicts. The grid and `/api/data` read the +same file (contract invariant I5, "one question, one path"). + +For a cheap entry it calls `ensure_materialized` and posts the entry's own +expanded build. A cheap build is a view in the database sense: a stored +definition (filter these rows, keep these columns) over files that exist. +Buckaroo's stats, sorts and pages run through it. Tallyman has already executed +that plan once, in full, at build time, so an error in it has already surfaced +in tallyman. + +Paddy confirmed this reading on 2026-09-20 (question 4 of the grilling +session). The words that decide it are "materialize the parquet if necessary": +a file is written when the entry is created and only for a worthy entry, and +nothing is written when an entry is viewed. The alternative, writing a file the +first time a cheap entry is opened so that Buckaroo only ever reads one file, +was set aside. It costs a wait on first open and a full copy per viewed +revision; the audit measured 19 GB of cache against 779 MB of data when every +CSV revision wrote a copy. + +The timing half of the rule already holds for creation. The build executes the +entry once before it writes the manifest, a worthy entry's snapshot is written +in that step (D4), and an entry with no manifest is treated as absent, so +Buckaroo cannot be asked to display an entry whose query is still running. The +same ordering now covers a snapshot that was deleted later: `load_session` +waits for `ensure_materialized` before it posts anything. + +Deleting a snapshot (the Cache page, a reset prune, a future budget eviction) +ends every live Buckaroo session whose plan reads it first, using the +session-eviction hook that ADR-006 decision D10 introduced. The next +`/api/session` re-materializes and opens a new session. Without this, a page +request against a deleted file would fail inside Buckaroo with the +`At least one path is required` error above, which is exactly the kind of +failure the rule says belongs to tallyman. + +This resembles what #104 removed: #102's viewer build over +`/result.parquet`. #104's objection was two materialized copies per +entry and two read paths. Here there is one copy and the view build only points +at it. + +*Rejected:* post the entry's own build for worthy entries too. With no cache +node in it, Buckaroo would re-run the full computation for every page and stat +query. + +### D7. The cold state is an empty `compute_cache` + +The contract's cold seam today is the `cache_dir` argument, and the standing +tests exercise it by deleting `compute_cache/` and reading again +(`tests/test_lineage_faithful_reads.py`). That remains the test: with +`compute_cache/` removed, the canonical read must reproduce every snapshot the +entry needs, each with its recorded digest. The `cache_dir` parameter leaves +the contract, since nothing in a build resolves through it any more. + +### D8. A sentinel test keeps xorq's cache out of the path + +With `XORQ_CACHE_DIR` pointing at an empty sentinel directory, a build, a +chained child build, a view, an eviction and a heal must leave the sentinel +empty. The test fails on `main` today (finding 1), so it belongs in the +failing-tests commit, and it outlives this change as the check that no xorq +cache node has crept back in. + +### D9. One change, one rebuild + +Removing the cache node changes the hash of every worthy entry, and bare-read +chaining changes every child's. This lands together with ADR-008's change to +`tallyman_read_csv` and ADR-009's digest definition, behind a single corpus +rebuild, with the failing tests committed and seen red first. After the +rebuild, `~/.cache/xorq/result_cache` (14 GB) and the older leaks under +`~/.cache/xorq/parquet/` (2.6 GB) can be deleted by hand. + +### D10. Every diff is built as an entry before it is displayed + +Live and promoted diffs already build the same expression +(`build_compare_expr`). The live path posts it to Buckaroo unmaterialized +(`_build_compare_expr`, `app.py:418-441`), so the outer join runs again for +every page, sort and stat query in the diff grid, and a failure of the join +surfaces inside Buckaroo. The promote path writes a recipe that calls +`build_diff_expr(a_hash, b_hash, keys)` and runs the normal build. + +Every diff now takes the promote path. When a diff view is opened, tallyman +builds the diff entry and waits for the build to finish. A diff contains a +join, so it is worthy and is materialized, and Buckaroo is then handed a view +build of the finished file (D6) with the diff's display configuration. The page +shows a "building diff" state while it waits. Promoting a diff is reduced to +putting an alias on an entry that already exists. + +- **An unnamed diff entry is ephemeral.** The checkpoint zips and commits every + complete directory under `entries/`, aliased or not (`zip_pending_entries`, + `catalog.py:160-180`), so an unnamed diff stored there would put every diff + ever viewed into the catalog's git history. Ephemeral entries live under + `compute_cache/ephemeral_entries//`, where everything is + already defined as deletable at any time, and they can be rebuilt from the + two hashes and the keys. Promote moves the directory into `entries/`, sets + the alias and checkpoints. The content hash is computed from the graph, so + the move does not change it, and the snapshot path stays the same. +- **Parents.** A worthy side is read from its snapshot. A cheap side is not + copied first: because the diff is itself materialized, each side is read + exactly once, while the diff is written. +- **Sessions.** The same entry can be opened as a diff, with the diff display + classes, or as a plain entry, so the Buckaroo session key includes the view + kind and is no longer the content hash alone. +- **Retired with this:** `_build_compare_expr` and its temp build directory, + the separate diff-session bookkeeping (`diff_session_is_loaded`, + `mark_diff_session_loaded`), and `diff_stat_cache/`, since a diff's Buckaroo + stats become its entry's own. The audit finding that every recalc wipes all + of `diff_stat_cache/` goes with it. + +*Rejected:* keep posting the join and materialize nothing. It shows a first +page sooner, and it breaks the governing rule in both directions: Buckaroo runs +tallyman's join, repeatedly, and tallyman never learns whether it succeeded. + +## Consequences + +- **Retired:** the `.cache()` call and the source-read injection in + `rewrite_for_build`; `rewrite_cache_dirs`; `_cached_node_path`; + `manifest.snapshot_key` and `_assert_recorded_snapshot_key`; the `baked` / + `recompute` plan split keyed on a `CachedNode`; the thread-only `_heal_lock` + and the `(FileNotFoundError, ValueError)` retry around xorq's shared temp + file; `entry_graph_expr` as a separate function. +- **ADR-006:** its D4 (inlined chaining) and D8 (snapshot-key tripwire) are + superseded. Its D2's "snapshot path derived from the loaded expression" + becomes `snapshot_path`. Its D3 (rebind composition onto the default backend) + is still needed for cheap parents and for diff composition. Its D5 (canonical + sort) and D6 (a missing build is a hard error) are unchanged. Its D7, D10 and + D12 (loud verification, the Buckaroo-state wipe, the pin and badge) attach to + `ensure_materialized`. +- **`docs/system-contract.md`** needs rewriting in Part 1 §4 (xorq caching + becomes background, not mechanism), "Content hash" (no cache injection in the + hashed expression; parents appear as snapshot paths), "Manifest" (drop + `snapshot_key`), "Worthiness", write path steps 2 and 5, the read path, + "Chaining", and the `[^preheal]` footnote, which currently describes + bare-read chaining as a retired workaround. +- **`plans/remove-ondemand-result-parquet.md`:** its premise that xorq's + snapshot is the single materialized copy is replaced. Tallyman's snapshot is + the single copy. +- **ADR-003:** chained descendants of an expensive entry are cheap, so its + motivating lineage stops filling the cache when the iterations chain off the + join. A revision that restates the join in its own recipe is still worthy + and still writes a copy, so the budget and eviction half of ADR-003 remains + and should be rewritten against this design. +- **CSV lineages:** a child of a `tallyman_read_csv` entry reads the parent's + snapshot and no longer inherits its Sort, so revisions stop baking one full + sorted copy each. The root entry's own copy is ADR-008's subject. +- **Buckaroo's stat keys** for a worthy entry become a function of one read of + one content-addressed path. +- **Source identity `salt` mode:** `rewrite_for_build` returns early under + `salt` because xorq's path-only snapshot keys would collide. A snapshot named + by the entry's content hash has no such collision, so the early return should + become unnecessary. Not tested. +- **Cost accepted:** a worthy parent's snapshot must exist before a child can + be built. A build also no longer repairs its own ancestors when something + outside tallyman executes it, and under the governing rule nothing does. + +## Open questions + +1. **Deep cheap chains.** Nothing cuts the graph between cheap entries. Run + `tests/test_perf_chain_depth.py` against this design and decide whether a + node-count or `compile_seconds` threshold should make an otherwise cheap + entry worthy. +2. **An unfaithful parent's descendants.** Descendants of an entry whose heal + failed verification were computed from bytes that no longer exist. Nothing + flags them today, and nothing here does either. +3. **Eviction policy.** D6 says what eviction must do to live sessions. Which + snapshots to evict, and when, stays with the ADR-003 rewrite. +4. **Garbage collection of ephemeral entries.** D10 says where they live and + that they are deletable. When to delete them belongs with the eviction + policy of open question 3. diff --git a/plans/ADR-008-row-order-of-reads.md b/plans/ADR-008-row-order-of-reads.md new file mode 100644 index 0000000..a7e0eea --- /dev/null +++ b/plans/ADR-008-row-order-of-reads.md @@ -0,0 +1,382 @@ +# ADR: Row order of reads (every file carries `__row_order`, every page sorts by it) + +- **Status:** Proposed (2026-09-18, revised 2026-09-20 in the grilling + session). The first draft pinned row order with an engine setting. Paddy + proposed baking a row-order column into every file tallyman writes and + sorting every page by it. The measurements below favour that, so it is now + the decision and the engine setting is the rejected alternative under D5. + Amends `plans/ADR-005-intelligent-csv-import.md` INV-1 (the name and position + of the row-order column) and INV-2 (the trailing `order_by`). Corrects a + threshold quoted in `plans/ADR-006-read-path-loads-builds.md` decision D5 + (the canonical sort) and three other places. +- **Context:** the 2026-09-18 cache audit (tallyman @ `a748ea6`, buckaroo + 0.15.4, xorq 0.3.26, xorq-datafusion 0.2.7). No ticket filed yet. +- **Affected code:** `src/tallyman_xorq/source_cache.py` (`rewrite_for_build`, + `_tie_break_order`), `src/tallyman_xorq/io.py` (`read_project_file`, + `tallyman_read_csv`, `io.py:627`), `src/tallyman_xorq/result_cache.py` + (`_EXPENSIVE_OPS`, `classify_build`), `src/tallyman_companion/app.py` + (`api_data`, `app.py:930`, and the chart data it feeds), + `src/tallyman_xorq/primary_key.py` (candidate selection, + `primary_key.py:219`), `src/tallyman_companion/diff.py` + (`build_compare_expr`), and the `materialize` writer introduced by + `plans/ADR-007-tallyman-owned-materialization.md` decision D4. Buckaroo's + paging is Buckaroo's code and is covered by D8. +- **Related ADRs:** `plans/ADR-004-result-digest-canonical-ordering.md` (why a + canonical stored order exists), `plans/ADR-007-tallyman-owned-materialization.md` + (what a snapshot is, and the corpus rebuild this shares), + `plans/ADR-009-digest-stability.md` (the file format, which D5 adds a + requirement to). +- **Evidence:** `scripts/spike_row_order_paging.py` (the decisions) and + `scripts/spike_window_read_order.py` (the problem, and the rejected + engine-setting approach). All figures are from those scripts on a 14-core + machine. + +## Terms + +- **Entry:** one catalog computation, stored under its content hash. +- **Materialize:** run an entry's computation once and write the result to a + parquet file. That file is the entry's **snapshot**. +- **Worthy entry:** an entry tallyman materializes. **Cheap entry:** one it + does not, whose small plan re-runs on every read. D4 draws the line. +- **Page request:** a request for `limit` rows starting at `offset`. `/api/data`, + charts and Buckaroo's grid all issue them. The first draft of this ADR called + it a "window". +- **Row-preserving:** each output row comes from exactly one input row, and no + input row produces more than one output row. Filters, column selections, + computed columns, renames and casts qualify. Aggregates, joins, unions, + distincts and unnests do not. +- **Tie:** two or more rows with equal values in every sort key. +- **Exchange operator:** a DataFusion physical-plan step (`RepartitionExec`, + `CoalescePartitionsExec`) that moves rows between parallel partitions. After + one, rows arrive in whatever order the partitions finish. + +## Problem + +Decision D5 of ADR-006 (the canonical sort) made the write deterministic: a +worthy entry's snapshot is written in a fixed total order, so the file is the +same on every rebuild. Nothing made the read of that file deterministic. +`/api/data` serves a page as +`cached_result_expr(project, hash).limit(limit, offset=offset).execute()` +(`app.py:930`), charts pull `limit=100000` through the same endpoint, and +Buckaroo pages the grid the same way in its own process. A `LIMIT/OFFSET` with +no `ORDER BY` takes rows in whatever order the plan delivers them. + +Eight identical requests for 50 rows from a 91 MB parquet file whose rows are +physically sorted by `id` (`scripts/spike_window_read_order.py`): + +| Plan | Offset | Distinct pages out of 8 | First id seen (file order would give) | +| --- | --- | --- | --- | +| read, limit | 0 | 5 | 0, 434176, 1294336 (0) | +| read, limit | 1,000,000 | 8 | 1360448, 1368640, 1565248 (1000000) | +| read, filter, computed column, limit | 0 | 8 | 1, 221185, 647168 (1) | +| read, filter, computed column, limit | 1,000,000 | 8 | 746336, 754526, 967522 (1500001) | + +A sort does not help when its key has ties. Sorting 3,000,000 rows by a column +with 200 distinct values and asking for the page at offset 100,000 returned 6 +different pages for 6 identical requests (`scripts/spike_row_order_paging.py`). +Sorting the grid by a category or a date is exactly that case. + +Paging through a large entry therefore repeats some rows and never shows +others, a chart over more than 100,000 rows changes each time it mounts, and an +author's own `order_by` is not honoured on screen: the file is sorted and the +page taken from it is not. + +**The governing variable** for the unsorted case is whether the physical plan +has an exchange operator between the scan and the limit. There are three ways +to get one: + +- DataFusion splits a file scan into byte ranges, one per partition, when the + file is larger than `datafusion.optimizer.repartition_file_min_size`. In this + engine that is 10,485,760 bytes (`SHOW ...` on a fresh `xo.connect()`), not + the 1 MiB the repo states. +- Above a single-partition scan the planner still inserts + `RepartitionExec: RoundRobinBatch(14)` under a filter or projection. +- A hash aggregate or join repartitions by key. This, and not file splitting, + is what the #171 probe observed: its fixture is a 2.07 MB file, below the + split threshold, and its plan is an aggregate. + +**What INV-2 of ADR-005 was for.** INV-2 keeps a trailing +`order_by("original_row_order")` on every `tallyman_read_csv` expression so +that the entry is worthy and its order is canonical. Its costs, measured in the +audit: + +- Every entry in a CSV lineage is worthy for that Sort alone, so every revision + writes a full sorted copy. A 9.3 MB CSV with four trivial revisions produced + 37 MB of snapshots. One real project holds 779 MB of data and 19 GB of + `compute_cache`, with ten snapshots of 0.45 to 3.7 GB that are all + `why=ops:Sort,SortKey`. +- Primary-key inheritance is gated on `not cache_worthy` + (`primary_key.py:204`), so it never applies in a CSV lineage and every + revision pays a full-table distinct scan. +- Above 10 MB it does not deliver a stable order on screen, for the reason + above. + +## Decisions + +### D1. The contract: a page is a function of `(content_hash, sort, offset, limit)` + +The same page request returns the same rows in the same order, in any process +and any cache state, with or without a user sort. With no user sort the rows +come in `__row_order` order (D2). This is the system contract's invariant I1 +("a content hash names a fixed result") applied to a page. + +### D2. Every file tallyman reads carries `__row_order` + +`__row_order` is an `int64` column holding `0..N-1` in the file's physical row +order. It is the last column, and it is visible: Buckaroo shows it as the final +column of the table. + +Two writers produce it: + +- **`materialize`** (ADR-007 decision D4, the one writer of snapshots) numbers + the rows of the canonically sorted stream as it writes them. If the stream + already has a `__row_order` inherited from a parent, the writer replaces that + one column. Each materialization therefore overwrites `__row_order` with + positions in its own file. +- **Ingest.** A source file enters tallyman through an ordered copy: polars + scans it, `with_row_index` numbers the rows in file order, and the copy is + written with the column last. CSVs already work this way (the intermediate + parquet under `csv_ordered/`). Parquet sources gain the same step, keyed by + the source's digest, beside the content-addressed clone that stays the + immutable input. A source that already has a `__row_order` column has it + overwritten, which is the right outcome for a file tallyman exported. + +The canonical sort's tie-break (`_tie_break_order`) puts an inherited +`__row_order` where `original_row_order` is today: after the author's own +`order_by` keys and before the remaining columns. A worthy entry that keeps its +parent's rows, such as one adding a window function, therefore keeps the +parent's order. + +*Rejected:* `row_number()` inside the entry's graph. It needs the same global +sort, adds a window function to every worthy build, and leaves contiguity to +the engine. A counter in the writer is contiguous and physical by construction, +which D5's range requests depend on. +*Rejected:* a row number kept only as file metadata. DataFusion exposes no +parquet row number that a query can sort or filter by, so it has to be a column. + +### D3. A cheap entry that drops `__row_order` is a build error + +`__row_order` is metadata that happens to be a column, and it is not to be +deleted. A column selection is an allow-list: `t.select("g", "n")` drops every +column it does not name, so an author who is not thinking about row order +drops it without meaning to. When a cheap entry's output lacks `__row_order`, +the build fails, and the error names the parent entry, says that a +row-preserving view must keep `__row_order` so that paging stays repeatable, +and shows the fix (`t.select("g", "n", "__row_order")`). The MCP tool +descriptions say the same thing up front. This is the feedback channel +`tallyman_read_csv` already uses when a CSV has a column named +`original_row_order`. + +A worthy entry is exempt, because the writer numbers its rows (D2). An author +changes `__row_order` by asking for an order: `order_by` makes the entry worthy, +and the writer numbers the rows in the requested order. Assigning to the column +directly stays an error (D6). + +Tallyman makes one alteration of its own, at the top of the expression only: it +moves `__row_order` to the last position, since a computed column added after +it would otherwise push it into the middle of the table. + +The column can be copied for debugging. +`foo_v1.mutate(__row_order_v1=foo_v1["__row_order"])` gives the new entry both +columns. Once the new entry is materialized, `__row_order` holds its own +positions and `__row_order_v1` still says where each row sat in the parent. + +*Rejected:* carry the column automatically. `rewrite_for_build` can rewrite +every column selection in a cheap graph to keep it, and the spike shows that +working: `t.filter(t.g < 100).select("g", "v0").mutate(z=t.v0 * 2)` comes out +with columns `['g', 'v0', 'z', '__row_order']` and pages repeatably. It was +rejected because the recipe text and the entry's columns would then disagree, +because it is surgery inside an expression an LLM wrote, and because a mistake +the author can fix in one line is better reported than silently repaired. The +cost accepted is a failed build whenever a select list forgets the column. +*Rejected:* carry the column and hide it from the grid. Paddy's call: it is +shown, as the final column. + +### D4. Cheap means row-preserving over one file; everything else is materialized + +A cheap entry inherits its row order, so it must be a row-preserving plan over +exactly one file. The classifier changes from a deny-list to an allow-list: an +entry is cheap only if every relation operation in its graph is known to be +row-preserving (a file read, a filter, a column selection, a computed column, a +rename, a cast, a column drop). Anything else is worthy, including operations +nobody has thought about yet. Today's `_EXPENSIVE_OPS` deny-list classes +`Union`, `Distinct` and `Unnest` as cheap, and none of them can carry one +parent's row order. + +A new or unknown operation now costs a copy (safe) instead of unstable paging +(unsafe). `classify_build` (which reads the serialized build) and +`_is_worthy_expr` (which reads the live expression) flip together, as they must +today. + +Supporting measurement from the first draft: a union of two files returned 2 +different pages for 8 identical requests even on a single-partition +connection, because `UnionExec` emits one partition per input and an exchange +operator merges them. + +### D5. Every page request orders by `__row_order` + +- No user sort: `ORDER BY __row_order`. +- User sort: the user's keys, then `__row_order` ascending as the last key, + which breaks every tie. + +That is the whole rule, for every entry and both processes. Two faster paths +exist and are deliberately not part of this decision (Paddy, 2026-09-20: a +cohesive system that works reliably comes first, and speed problems are handled +as they come up): + +- **Declared file order.** Telling the engine the file is already sorted by + `__row_order` removes the sort from the plan. +- **Range request.** For the unfiltered, unsorted view of a materialized file a + row's position equals its `__row_order`, so a page can be fetched as + `__row_order >= offset AND __row_order < offset + limit` instead of `OFFSET`. + +Both are measured below so the numbers are on hand when they are wanted. + +Measured on 3,000,000 rows by 14 columns (287 MB), on the default parallel +connection with no engine settings. Every row of the table returned the correct +page 6 times out of 6: + +| Page request | Offset 0 | Offset 1,000,000 | Offset 2,900,000 | +| --- | --- | --- | --- | +| `ORDER BY __row_order LIMIT 50 OFFSET k` | 100 ms | 332 ms | 374 ms | +| The same, with the file's order declared to the engine (`WITH ORDER`) | 25 ms | 102 ms | 242 ms | +| Range request, row-group statistics only | 90 ms | 90 ms | 79 ms | +| Range request, file written with a parquet page index | 19 ms | 24 ms | 23 ms | +| First draft's approach: bare `LIMIT/OFFSET`, single-partition connection | 21 ms | 90 ms | 249 ms | + +And the sorted case, at offset 100,000: `ORDER BY g` gave 6 distinct pages in 6 +requests (214 ms); `ORDER BY g, __row_order` gave 1 (302 ms). + +One consequence for the file format, recorded in ADR-009 decision D3: the +writer puts `__row_order` last. It also emits a parquet page index, which costs +nothing now and is what takes a later range request from 90 ms to 20 ms. + +Declaring the file's order to the engine removes the sort from the plan +(`GlobalLimitExec <- SortPreservingMergeExec <- DataSourceExec`, no +`SortExec`). The spike registers the file through +`CREATE EXTERNAL TABLE ... WITH ORDER`. Whether the declaration can travel +inside a xorq build is untested (open question 3), so it is an optimization +here and not part of the decision. + +*Rejected:* the first draft's decision, a second connection with +`target_partitions = 1` for page requests. It makes unsorted pages repeatable +(bottom row of the table) at the same cost as a declared order. It was +rejected because: + +- it depends on the engine's planner never introducing an exchange operator, + and `repartition_file_scans = false`, which looked sufficient in an earlier + experiment, returned the same wrong page 8 times out of 8 for a filtered plan + on a file with 100,000-row groups; +- it has to be reproduced inside Buckaroo's process; +- it does nothing for a user sort with ties; +- it fails for a union. + +An `ORDER BY` on a column with no ties is repeatable by the query's own +semantics, in any engine and any process. + +### D6. The exact name `__row_order` is reserved + +- A recipe may read the column and may copy it under another name (D3). A + recipe that assigns to `__row_order` is a build error, because arbitrary + values could contain ties or gaps, and D5 depends on `0..N-1` with neither. +- Only the exact name is special. `__row_order_v1`, or any other name an author + picks for a copy, is ordinary data and survives materialization. +- A join of two entries leaves the right side's copy behind under ibis's + collision name, `__row_order_right`, and a three-way join silently keeps only + the first two. That column is ordinary data too: it says where the row sat in + the right-hand parent. The writer replaces only `__row_order` itself. +- The primary-key search skips it. Nothing excludes `original_row_order` from + the candidates today (`primary_key.py:219`), and a column that is unique in + every table would win the search for any table without a string or id key. + Row positions shift between versions, so a diff keyed on it would be + meaningless. +- `build_compare_expr` drops it from both sides before joining. The diff is an + entry (ADR-007 decision D10) and gets its own when it is materialized. + +### D7. `tallyman_read_csv` loses its trailing `order_by`, and its column becomes `__row_order` + +Amends ADR-005. INV-2: `io.py:627` returns a plain read of the intermediate +parquet with no `order_by`. INV-1: the row-index column is named `__row_order` +and written last, so a CSV root has one row-order column and not two with +identical values. The root entry becomes a cheap read of the intermediate, and +D5 gives file-order pages with no Sort and no second copy. + +What INV-2 provided, and what replaces it: + +| INV-2 gave | Replacement | +| --- | --- | +| A canonical display order | D5. INV-2 did not deliver this above 10 MB. | +| A parquet boundary for chained children | A cheap root's graph is one read node. | +| A `result_digest` on the root, so a re-parse that produced different rows would be caught | Lost as it stands: cheap entries record no digest (ADR-006 decision D9, "no cheap-entry digests"). See open question 5. | + +Every hash in every CSV lineage changes, so this rides the corpus rebuild of +ADR-007 decision D9 ("one change, one rebuild"). + +### D8. Buckaroo's half is one hint and one Buckaroo issue + +Buckaroo pages in its own process, so the grid needs the same rule and tallyman +cannot apply it from outside. Tallyman passes the column's name in the +`/load_expr` payload as a hint. The Buckaroo issue asks that, given the hint, +Buckaroo sorts by it when the user has chosen no sort, appends it as the last +key of any user sort, and may use range requests for the unfiltered, unsorted +view. Without the hint Buckaroo behaves as it does now. Filed as +buckaroo-data/buckaroo#974. + +The issue reproduces the defect through Buckaroo's own page builder. +`_window_to_parquet` (`buckaroo/xorq_buckaroo.py`, lines 302-330 in 0.15.6) +sorts on a single key, `expr.order_by(expr[sort_col].asc())`, and applies no +`order_by` at all when the user has chosen no sort. Six identical calls for the +same 50 rows of a 27 MB file gave 5 different results unsorted and 6 sorted by +a column with 200 distinct values. + +### D9. Correct the threshold + +`repartition_file_min_size` is 10,485,760 in this engine. Four places say 1 MiB +and are corrected with this change: `plans/ADR-006-read-path-loads-builds.md:98`, +`plans/datafusion-scan-order-findings.md:64`, +`tests/test_parquet_digest_order_probe.py:5` (whose `> 1_048_576` size guard +does not establish a split scan and is not what makes that test meaningful; +its aggregate is), and `src/tallyman_xorq/source_cache.py:98`. +`tests/test_tallyman_read_csv.py:159` already says about 10 MB. + +## Consequences + +- Pages are repeatable for unsorted and sorted requests, in tallyman and in + Buckaroo, with no engine settings and no second connection. +- Every table shows one more column, at the end. A join result also shows + `__row_order_right` unless the recipe drops it. +- A recipe whose select list forgets `__row_order` fails to build until the + author adds it. +- CSV lineages stop writing sorted copies. With ADR-007, revisions of a CSV + entry are cheap reads over one intermediate file, and primary-key inheritance + applies to them. +- Unions, distincts and unnests are materialized, which is the price of having + a defined row order. +- Each parquet source costs one ordered copy, about the size of the source. +- An unsorted page costs a sort of one column unless the file's order is + declared (100 to 374 ms against 25 to 242 ms in the spike). A range request + costs about 20 ms at any depth. +- The grid stays unstable until Buckaroo's half (buckaroo-data/buckaroo#974) + lands: above 10 MB when unsorted, and at any size when sorted by a column + with ties. + +## Open questions + +1. **Ordered copies of parquet sources.** D2 adds a copy per parquet source. + Adopted as the uniform rule under Paddy's "cohesive first" priority, and not + yet confirmed by him in so many words. It also assumes polars numbers a + parquet scan's rows in file order, as ADR-004 measured for CSV, which needs + checking. +2. **Renaming `original_row_order`.** D7 replaces it with `__row_order`. + Adopted on the same basis, and also not yet confirmed. The alternative keeps it as a data column meaning "line of + the source file", at the cost of two identical columns on every CSV root. +3. **Declaring a file's order inside a xorq build.** It works through DDL on a + connection. If it can ride in a build's read node, Buckaroo's unsorted pages + get the cheaper plan too. +4. **Deep offsets on a cheap entry.** Positions in a filtered view have gaps, + so it pages with `OFFSET`, whose cost grows with depth. +5. **A digest for an ordered copy.** An ordered copy is the record of a parse + and is re-created from the clone if deleted. Recording its digest in the root + entry's manifest, and verifying it on re-creation, would restore what D7 + gives up. It wants ADR-009's digest definition, and it touches where the + copies live: `csv_ordered` is global, is never collected, and is not packed. diff --git a/plans/ADR-009-digest-stability.md b/plans/ADR-009-digest-stability.md new file mode 100644 index 0000000..5f9763c --- /dev/null +++ b/plans/ADR-009-digest-stability.md @@ -0,0 +1,268 @@ +# ADR: Digest stability (a heal is flagged only when the result changed) + +- **Status:** Proposed (2026-09-18, revised 2026-09-20: D3 gains two format + requirements from `plans/ADR-008-row-order-of-reads.md`, and D1 lost its + speed gate). Amends + `plans/ADR-004-result-digest-canonical-ordering.md` (Option A's "hash the + snapshot bytes") and decision D5 of + `plans/ADR-006-read-path-loads-builds.md` (the canonical sort), which said + "`result_digest` keeps its file-hash definition". The canonical sort itself + is unchanged and is still required. +- **Reading decision labels:** a bare label such as "D2" in this document + always means this ADR's own decision. Another ADR's decision is always + written with its ADR number and a few words saying what it decides. +- **Context:** the 2026-09-18 cache audit (tallyman @ `a748ea6`, xorq 0.3.26, + xorq-datafusion 0.2.7, pyarrow 21.0.0). No ticket filed yet. +- **Affected code:** `src/tallyman_xorq/result_cache.py` + (`snapshot_file_digest`, `verify_result_faithful`, `_verify_self_heal`), + `src/tallyman_xorq/build.py` (the execute-once step), + `src/tallyman_xorq/backend.py` (a single-partition connection), + `src/tallyman_core/manifest.py` (`result_digest`), and the `materialize` + writer that decision D4 of `plans/ADR-007-tallyman-owned-materialization.md` + introduces (one writer for snapshots, used by the build and by every heal). + D2 and D3 assume that writer. If ADR-007 were rejected, D1 would stand as + written and D2 would need restating against xorq's writer. +- **Related ADRs:** `plans/ADR-008-row-order-of-reads.md` (uses the same + single-partition setting for a different job, and has an open question this + digest would answer). +- **Evidence:** `scripts/spike_float_aggregate_digest.py`, + `scripts/spike_logical_digest.py`. + +## Terms + +- **Materialize:** run an entry's computation once and write the result to a + parquet file. That file is the entry's **snapshot**. +- **Worthy entry:** an entry tallyman materializes. A cheap entry is one it + does not, and it records no digest. +- **`result_digest`:** the value recorded in a worthy entry's manifest when its + snapshot is first written, against which every later rewrite is checked. +- **Heal:** re-create a snapshot that is missing from disk by re-running the + entry's build. +- **Unfaithful heal:** a heal whose digest does not match the recorded one. +- **Canonical sort:** the fixed total order in which a snapshot's rows are + written (the author's `order_by` keys, then an inherited `__row_order`, then + the remaining columns), from ADR-006 decision D5. +- **Partition:** one of the parallel streams DataFusion splits a query into. + `target_partitions = 1` runs a query as a single stream. + +## Problem + +`result_digest` is the SHA-256 of the snapshot file's bytes. The contract gives +it one job: witnessing that a later rematerialization reproduced the original. +When a heal's digest does not match, tallyman writes a durable +`unfaithful_heal` record, pushes an SSE event, wipes the entry's Buckaroo stat +cache, evicts its session, and pins and badges the entry (ADR-006 decisions D7, +loud verification; D10, the Buckaroo-state wipe; and D12, the pin and badge). +That response is right for a recipe that calls `sample()`. It is +expensive, sticky and misleading when nothing about the result changed, and +today two things trigger it without a change in the result. + +**1. A float aggregate is not bit-reproducible under parallel execution.** +DataFusion sums each partition separately and merges the partial sums in +arrival order. Float addition is not associative, so the low bits move from +run to run. A group-by over 3,000,000 rows with a float `SUM` and `AVG`, +canonically ordered, six runs per configuration: + +| Aggregate | Partitions | Distinct digests in 6 runs | Median | +| --- | --- | --- | --- | +| float | default (14) | 6 | 6 ms | +| float | `target_partitions = 1` | 1 | 20 ms | +| integer | default (14) | 1 | 7 ms | +| integer | `target_partitions = 1` | 1 | 31 ms | + +The largest relative difference between two float runs was 6.4e-16. At the +tallyman level the audit healed a float-aggregate entry six times: six +different files, six `unfaithful_heal` records, six stat-cache wipes. An +integer-only aggregate healed byte-identical six times. The canonical sort +cannot help, since the rows are in the same order and it is the values that +differ. Aggregate is the usual reason an entry is worthy, so most expensive +entries with a float measure are flagged after any eviction. The message +blames "execution (#83), a fixed graph that runs differently each execute", +which sends the author to a recipe that is deterministic. The default partition +count is also the machine's core count, so the same build merges differently +on a different machine. + +**2. File bytes depend on things that are not the result.** + +- How the stream was batched. The same rows handed to a parquet writer as + 100,000-row batches and as 8,192-row batches gave different files unless each + row group was first combined into contiguous arrays. +- The writer's version. The footer carries + `created_by = 'parquet-cpp-arrow version 21.0.0'`, so after a pyarrow upgrade + every heal would mismatch. +- Under xorq's writer today, whether the cache node was the root of the + executed expression (1,234 bytes against 858 for the same rows). + +ADR-004 accepted "a library upgrade changes the hash; rebuild is fine". That +was written before a mismatch became loud. With those three ADR-006 decisions +in place, an +upgrade would pin and badge every entry that gets evicted until the corpus is +rebuilt. + +## Decisions + +### D1. Materialization runs single-partition + +`materialize` (ADR-007 decision D4, the one writer of snapshots) executes on a +connection configured with +`SET datafusion.execution.target_partitions = 1`. The build and every heal use +it, so both run one plan with one merge order, on any machine. It is a +different connection object from ADR-008's window connection, with the same +setting, because a long materialization must not share a context with page +reads. + +Cost: about 3x on the spike's aggregate. ADR-004 measured a 3.1M-group +aggregate at 0.5 s parallel against 3.4 to 3.9 s single-partition, and a full +43-column read at 2.5 s against 6.9 s, on the 11.8M-row parking file. A +materialization runs once per entry and once per heal, never on a read. + +ADR-004 rejected pinning scan order through session config as the *ingest* +lever, because it serializes everything downstream. This decision uses the same +setting for a different job and accepts that cost knowingly: reproducible +arithmetic is the point, and no setting gives both. + +Paddy's call in the grilling session (2026-09-20): "I want a cohesive system +that works reliably, then we can worry about speed problems as they come up." +So every materialization runs single-partition, and this decision carries no +speed gate. If the cost becomes a problem, the known variant is to run +single-partition only for a plan with a floating-point reduction or window, +decided once at build from the expression and recorded in the manifest so that +a heal never re-derives it. + +*Rejected:* round floats before hashing. A value next to a rounding boundary +flips under one unit of noise in the last place, and among millions of values +some are. +*Rejected:* compare with a tolerance. When a heal runs the original values are +gone and only the digest is left. +*Rejected:* treat a mismatch as advisory when the plan has a float aggregate. +That removes verification from the entries most likely to be expensive. +*Rejected:* drop the canonical sort, since single-partition output is ordered +anyway. The sort is what puts the author's keys first in the order the rows +are numbered (`__row_order`, ADR-008 decision D2), and it keeps the result independent of the hash +table's emission order, which an engine upgrade can change. + +### D2. `result_digest` is a digest of the snapshot's content, computed from the file as read back + +The digest is a SHA-256 over the snapshot's ordered Arrow data: + +- one hash stream per column for validity (one byte per row), one for lengths + (variable-width types) and one for values, with null slots zeroed or emptied, + since a null slot may hold anything; +- each stream seeded with the column name and its logical type, where `string`, + `large_string` and `string_view` are one type; +- the streams combined in schema order together with the row count; +- stored with an algorithm prefix (`arrow-sha256:`) so a future definition can + never be compared against this one by accident. + +Separate streams matter. The first version of the spike fed validity and +values into one hasher per column, and the digest then depended on where batch +boundaries fell. + +One function computes it, from the written file read back, for both the build +and verify. This is the "one derivation route" rule behind ADR-006 decision D8 +(the snapshot-key tripwire: the build and the read share one derivation). The +spike shows why it cannot be computed from the stream handed to the writer: the +writer coerced a `timestamp[s]` column to `timestamp[ms]`, and the two digests +differed. + +Measured on 3,000,000 rows with nulls, NaNs, strings, booleans and timestamps: + +| Check | Result | +| --- | --- | +| Same rows as 8,192-row, 100,000-row and single batches | 1 digest | +| Same file read back in 1,000-row batches | same | +| Same rows written Snappy with 8,192-row groups instead of zstd with 1,048,576 | same | +| Same file read through DataFusion with `order_by(id)` | same | +| One value changed | differs | +| One null replaced by `0.0` | differs | +| First two rows swapped | differs | +| Read back and hash a 57 MB file (127 MB of Arrow data) | 0.12 s, against 0.02 s for a file-bytes hash | + +The digest stays order-sensitive, so the canonical sort is still what makes it +reproducible. `__row_order` is a column of the file like any other, so the +digest covers it. + +This reverses part of the reasoning in ADR-006 decision D5 (the canonical +sort). That decision rejected the multiset digest +partly because "verify must read every row and the digest stops being a hash +of the artifact". Both are true of this digest as well. They are accepted here +because verify runs only on a heal and in the sweep, at roughly 1 GB/s of Arrow +data, and because being a hash of the artifact is exactly what turns a writer +upgrade into a false alarm. + +*Rejected:* keep file bytes and pin the writer's settings. It is reproducible +today (the spike gets one digest across batch sizes once each row group is +combined) and six times cheaper to verify. It also freezes the codec and +row-group size for the life of the corpus, and it fails on the first pyarrow +upgrade. +*Rejected:* polars `hash_rows`. Its documentation does not guarantee stable +results across polars versions. + +### D3. The snapshot's format + +Tallyman's writer (ADR-007 decision D4) regroups the record-batch stream into +row groups of 1,048,576 rows, combines each row group into contiguous arrays, +and writes zstd level 3, parquet format 2.6, statistics on. For the spike's rows that is +57.1 MB, 3 row groups and a 2.7 KB footer, against 100.3 MB, 367 row groups +and a 228.5 KB footer in the shape xorq writes (one Snappy row group per +8,192-row batch). Memory is bounded by one row group. + +ADR-008 adds two requirements. The writer numbers the rows in a last column +named `__row_order` (ADR-008 decision D2). And it writes a parquet page index, +which is what lets a page be fetched as a range of `__row_order` values without +decoding a whole row group: 19 to 24 ms at any depth with the index against 79 +to 90 ms without it, on a 287 MB file (`scripts/spike_row_order_paging.py` on +the ADR-008 branch). + +Combining each row group keeps the file bytes reproducible as well. Nothing +depends on that after D2, and it means two writes of the same entry can still +be compared with `cmp` when debugging. Because of D2 these settings can change later without +touching any digest. + +### D4. A mismatch record names its likely cause + +With D1 and D2 in place the causes left are the recipe's own nondeterminism +(`sample()`, `now()`, an impure UDF), source drift under `off` identity mode, +and an engine upgrade that changed results. The manifest records the xorq, +xorq-datafusion and pyarrow versions at build, and the `unfaithful_heal` record +carries them alongside the versions at heal. When they differ, the message says +so instead of blaming the recipe. The contract's attribution table gains that +row. + +### D5. It lands with the rebuild + +Every worthy entry's digest is recomputed by the corpus rebuild of ADR-007 +decision D9 (one change, one rebuild). +The manifest field keeps its name. Tests that cannot fail first ride with the +fix; the float-aggregate heal test and a batch-boundary digest test fail on +`main` and belong in the failing-tests commit. + +## Consequences + +- A float-aggregate entry heals to the same digest, and a pyarrow upgrade no + longer flags anything. The loud responses of ADR-006 decisions D10 and D12 + are left for the causes they were designed for. +- Materializations and heals are slower, by about 3x on aggregation at spike scale and up + to 7x in ADR-004's parking measurement. Reads are unaffected. +- Verify decodes the file instead of hashing its bytes. It runs on a heal and + in `catalog_scan_staleness(verify_results=True)`, never on a read. +- Snapshots are smaller, and their footers were 85 times smaller in the spike, + which matters to every page request that opens one. +- `docs/system-contract.md` changes in "Result digest" and in the manifest + table ("SHA-256 of the baked result snapshot"), and its verification table + gains the engine-change row. +- The same function can digest the CSV intermediate (ADR-008, open + question 1). + +## Open questions + +1. **Nested types.** The spike covers fixed-width, boolean, string and binary + columns. Lists, structs and maps need a recursive definition. +2. **Types parquet cannot store as given.** A `timestamp[s]` column comes back + as `timestamp[ms]`, so the snapshot's schema differs from the entry's + recorded schema. Either the writer refuses such a column, or the entry's + schema is recorded from the snapshot. `__row_order` pushes toward the second + answer, since the writer adds a column the entry's graph does not have. +3. **Engine upgrades.** Single-partition execution fixes the merge order within + one engine version. Nothing guarantees float results across versions. D4 + attributes that case and the remedy stays a rebuild. diff --git a/scripts/spike_bare_read_chaining.py b/scripts/spike_bare_read_chaining.py new file mode 100644 index 0000000..cf5044e --- /dev/null +++ b/scripts/spike_bare_read_chaining.py @@ -0,0 +1,117 @@ +"""ADR-007 evidence: chaining through a bare read of the parent's snapshot, with no xorq cache node anywhere. + +Raw xorq plus tallyman's ``classify_build``; no tallyman build machinery. A worthy parent (an aggregate, canonically +sorted) is built with no cache node and materialized by a plain parquet write to ``/.parquet``. +A child reads that path with ``deferred_read_parquet`` and adds a filter and a computed column. + +Questions, in the order printed: + +1. What does the child's build hash depend on: the snapshot's path string, or its bytes/mtime? +2. Can a child be built while the parent's snapshot is absent? +3. Does ``load_expr`` of the child's build need the snapshot? What does executing without it raise? +4. After the parent is re-materialized from its own build, does the child execute to the right answer? +5. Is the child classified cheap? (Under today's inlined chaining the same child is worthy.) +6. Does the child's ``expr.yaml`` carry the literal path, so the portable-path placeholder applies to it? +7. Did anything get written under xorq's global cache directory? + + uv run python scripts/spike_bare_read_chaining.py +""" + +from __future__ import annotations + +import hashlib +import os +import tempfile +from pathlib import Path + +HOME = Path(tempfile.mkdtemp(prefix="spike_bare_read_")) +os.environ["XORQ_CACHE_DIR"] = str(HOME / "_global_xorq") # must be set before xorq is imported + +import numpy as np # noqa: E402 +import pyarrow as pa # noqa: E402 +import pyarrow.parquet as pq # noqa: E402 +import xorq.api as xo # noqa: E402 +from xorq.ibis_yaml.compiler import build_expr, load_expr # noqa: E402 + +from tallyman_xorq.result_cache import classify_build # noqa: E402 + +N = 400_000 +BUILDS = HOME / "builds" +SNAPSHOTS = HOME / "compute_cache" / "result_cache" + + +def materialize(expr, dest: Path) -> str: + tmp = dest.with_name(f".{dest.name}.{os.getpid()}.tmp") + pq.write_table(expr.to_pyarrow(), tmp, compression="zstd") + os.replace(tmp, dest) + return hashlib.sha256(dest.read_bytes()).hexdigest()[:16] + + +def child_of(snapshot: Path, schema=None): + parent = xo.deferred_read_parquet(str(snapshot), schema=schema) + return parent.filter(parent.n > 10).mutate(r=parent.s / parent.n) + + +def attempt(fn) -> str: + try: + return str(fn()) + except Exception as exc: # noqa: BLE001 - the spike reports whatever is raised + return f"raises {type(exc).__name__}: {str(exc)[:90]}" + + +def main() -> None: + SNAPSHOTS.mkdir(parents=True) + source = HOME / "cas" / "deadbeef.parquet" + source.parent.mkdir() + rng = np.random.default_rng(7) + pq.write_table( + pa.table({"id": np.arange(N), "g": rng.integers(0, 20_000, N), "v": rng.integers(0, 1000, N)}), source + ) + + t = xo.deferred_read_parquet(str(source)) + parent = t.group_by("g").agg(n=t.count(), s=t.v.sum()).order_by(["g", "n", "s"]) + parent_build = Path(build_expr(parent, builds_dir=BUILDS)) + snapshot = SNAPSHOTS / f"{parent_build.name}.parquet" + first_digest = materialize(load_expr(parent_build), snapshot) + print(f"parent {parent_build.name}: {classify_build(parent_build)}, snapshot digest {first_digest}\n") + + same_path = Path(build_expr(child_of(snapshot), builds_dir=BUILDS)).name + original = snapshot.read_bytes() + pq.write_table(pa.table({"g": [1, 2], "n": [99, 98], "s": [5, 6]}), snapshot) # other rows, other size and mtime + other_bytes = Path(build_expr(child_of(snapshot), builds_dir=BUILDS)).name + elsewhere = SNAPSHOTS / "0123456789ab.parquet" + elsewhere.write_bytes(original) + other_path = Path(build_expr(child_of(elsewhere), builds_dir=BUILDS)).name + print(f"1. child hash: {same_path}") + print(f" same path with other bytes: {other_bytes}; same bytes at another path: {other_path}") + + snapshot.unlink() + print(f"2. compose with snapshot absent, no schema: {attempt(lambda: child_of(snapshot).schema().names)}") + with_schema = attempt(lambda: build_expr(child_of(snapshot, parent.schema()), builds_dir=BUILDS)) + print(f" build with snapshot absent, schema given: {with_schema}") + + child_build = BUILDS / same_path + print(f"3. load_expr(child) with snapshot absent: {attempt(lambda: type(load_expr(child_build)).__name__)}") + print(f" execute with snapshot absent: {attempt(lambda: load_expr(child_build).count().execute())}") + + again = materialize(load_expr(parent_build), snapshot) + rows = int(load_expr(child_build).count().execute()) + want = int(parent.filter(parent.n > 10).count().execute()) + print(f"4. parent re-materialized from its build: digest unchanged={again == first_digest}") + print(f" child rows {rows}, expected {want}") + + inlined = Path(build_expr(parent.filter(parent.n > 10).mutate(r=parent.s / parent.n), builds_dir=BUILDS)) + print(f"5. classify_build(bare-read child) = {classify_build(child_build)}") + print(f" classify_build(same child, parent graph inlined) = {classify_build(inlined)}") + + yaml_text = (child_build / "expr.yaml").read_text() + inlined_size = len((inlined / "expr.yaml").read_text()) + print(f"6. literal snapshot path in child expr.yaml: {str(snapshot) in yaml_text}") + print(f" child expr.yaml is {len(yaml_text):,} bytes; with the parent graph inlined it is {inlined_size:,}") + + leaked = sorted(str(f) for f in (HOME / "_global_xorq").rglob("*.parquet")) + print(f"7. parquet files under XORQ_CACHE_DIR: {leaked or 'none'}") + + +if __name__ == "__main__": + main() diff --git a/scripts/spike_float_aggregate_digest.py b/scripts/spike_float_aggregate_digest.py new file mode 100644 index 0000000..55f2c67 --- /dev/null +++ b/scripts/spike_float_aggregate_digest.py @@ -0,0 +1,77 @@ +"""ADR-009 evidence: a float aggregate is not bit-reproducible under DataFusion's parallel partial aggregation. + +Runs the same canonically ordered group-by six times per configuration and hashes the result's Arrow buffers. +A float ``SUM``/``AVG`` comes back with different low bits from run to run under the default partition count, and +with identical bits under ``target_partitions = 1``. An integer-only aggregate is the control: it is identical in +both configurations. Median wall time is printed so the cost of single-partition execution is visible. + + uv run python scripts/spike_float_aggregate_digest.py +""" + +from __future__ import annotations + +import hashlib +import statistics +import tempfile +import time +from pathlib import Path + +import numpy as np +import pyarrow as pa +import pyarrow.parquet as pq +import xorq.api as xo + +N = 3_000_000 +RUNS = 6 + + +def value_digest(table: pa.Table) -> str: + h = hashlib.sha256() + for col in table.columns: + for chunk in col.chunks: + for buf in chunk.buffers(): + if buf is not None: + h.update(buf) + return h.hexdigest()[:12] + + +def run(path: Path, partitions: int | None, kind: str) -> tuple[str, float, pa.Table]: + con = xo.connect() + if partitions is not None: + con.raw_sql(f"SET datafusion.execution.target_partitions = {partitions}") + t = con.read_parquet(str(path)) + if kind == "float": + expr = t.group_by("g").agg(n=t.count(), s=t.v.sum(), m=t.v.mean()).order_by("g") + else: + expr = t.group_by("g").agg(n=t.count(), s=t.k.sum(), hi=t.k.max()).order_by("g") + t0 = time.perf_counter() + table = expr.to_pyarrow() + return value_digest(table), time.perf_counter() - t0, table + + +def main() -> None: + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "t.parquet" + rng = np.random.default_rng(5) + source = pa.table( + {"g": rng.integers(0, 1_000, N), "v": rng.random(N) * 1e6, "k": rng.integers(0, 1_000_000, N)} + ) + pq.write_table(source, path, row_group_size=100_000) + groups = pq.ParquetFile(path).metadata.num_row_groups + print(f"source: {N:,} rows, {groups} row groups, {path.stat().st_size / 1e6:.1f} MB\n") + for kind in ("float", "int"): + for label, partitions in (("default partitions", None), ("target_partitions=1", 1)): + results = [run(path, partitions, kind) for _ in range(RUNS)] + digests = [d for d, _, _ in results] + median_ms = statistics.median(s for _, s, _ in results) * 1000 + distinct = len(set(digests)) + line = f"{kind:5s} {label:20s} distinct digests over {RUNS} runs: {distinct}, median {median_ms:.0f} ms" + if kind == "float" and len(set(digests)) > 1: + a, b = results[0][2].column("s").to_numpy(), results[1][2].column("s").to_numpy() + rel = np.max(np.abs(a - b) / np.abs(a)) + line += f", max relative difference between two runs {rel:.1e}" + print(line) + + +if __name__ == "__main__": + main() diff --git a/scripts/spike_logical_digest.py b/scripts/spike_logical_digest.py new file mode 100644 index 0000000..7a695d1 --- /dev/null +++ b/scripts/spike_logical_digest.py @@ -0,0 +1,229 @@ +"""ADR-009 evidence: two candidate definitions of ``result_digest`` for a tallyman-written snapshot. + +**File bytes** (today's definition). Reproducible only if the writer's input and settings are pinned: the same rows +arriving in differently sized record batches must be regrouped into fixed row groups, and each row group combined +into contiguous arrays before it is written. The parquet footer also embeds the writer's version string, so the +bytes move on every pyarrow upgrade. + +**Logical content**. A SHA-256 over the ordered Arrow data, one hasher per column, fed a normalized form that does +not depend on batch boundaries, on the physical string type, or on whatever sits in null slots. It is independent +of codec, row-group size and writer version, and costs a read-back to verify. + +The script checks both for the invariances each needs, compares the file shape of a tallyman-side writer with +the shape xorq's ``ParquetStorage`` writes, then times the two digests. + + uv run python scripts/spike_logical_digest.py +""" + +from __future__ import annotations + +import hashlib +import tempfile +import time +from pathlib import Path + +import numpy as np +import pyarrow as pa +import pyarrow.compute as pc +import pyarrow.parquet as pq +import xorq.api as xo + +N = 3_000_000 +ROW_GROUP = 1_048_576 + + +def _logical_type(dtype: pa.DataType) -> str: + if pa.types.is_string(dtype) or pa.types.is_large_string(dtype) or pa.types.is_string_view(dtype): + return "string" + if pa.types.is_binary(dtype) or pa.types.is_large_binary(dtype) or pa.types.is_binary_view(dtype): + return "binary" + return str(dtype) + + +class LogicalDigest: + """Order-sensitive digest of a record-batch stream, invariant to chunking and physical encoding.""" + + # Each column keeps one hasher per byte stream (validity, lengths, values). A single hasher per column would + # interleave the streams chunk by chunk, and the digest would then depend on where the batch boundaries fall. + STREAMS = ("validity", "lengths", "values") + + def __init__(self, schema: pa.Schema): + self.schema = schema + self.hashers = [ + {s: hashlib.sha256(f"{f.name}\x00{_logical_type(f.type)}\x00{s}".encode()) for s in self.STREAMS} + for f in schema + ] + self.rows = 0 + + def update(self, batch: pa.RecordBatch | pa.Table) -> None: + for hashers, column in zip(self.hashers, batch.columns): + chunks = column.chunks if isinstance(column, pa.ChunkedArray) else [column] + for arr in chunks: + self._update_array(hashers, arr) + self.rows += batch.num_rows + + @staticmethod + def _update_array(hashers: dict, arr: pa.Array) -> None: + if pa.types.is_dictionary(arr.type): + arr = arr.dictionary_decode() + nulls = arr.is_null().to_numpy(zero_copy_only=False) + hashers["validity"].update(nulls.astype(np.uint8).tobytes()) # one byte per row, null or not + dtype = arr.type + if _logical_type(dtype) in ("string", "binary"): + target = pa.large_string() if _logical_type(dtype) == "string" else pa.large_binary() + filled = pc.fill_null(arr.cast(target), "" if target == pa.large_string() else b"") + hashers["lengths"].update( + pc.binary_length(filled).to_numpy(zero_copy_only=False).astype(np.int64).tobytes() + ) + offsets = np.frombuffer(filled.buffers()[1], dtype=np.int64)[ + filled.offset : filled.offset + len(filled) + 1 + ] + if len(filled): + hashers["values"].update(memoryview(filled.buffers()[2])[int(offsets[0]) : int(offsets[-1])]) + elif pa.types.is_boolean(dtype): + hashers["values"].update(pc.fill_null(arr, False).to_numpy(zero_copy_only=False).astype(np.uint8).tobytes()) + elif pa.types.is_primitive(dtype) or pa.types.is_decimal(dtype) or pa.types.is_fixed_size_binary(dtype): + width = dtype.bit_width // 8 + raw = np.frombuffer(arr.buffers()[1], dtype=np.uint8)[arr.offset * width : (arr.offset + len(arr)) * width] + if arr.null_count: + raw = raw.reshape(len(arr), width).copy() + raw[nulls] = 0 # null slots may hold anything; zero them + hashers["values"].update(raw.tobytes() if arr.null_count else memoryview(raw)) + else: + raise NotImplementedError(f"nested/other type not covered by the spike: {dtype}") + + def hexdigest(self) -> str: + top = hashlib.sha256(str(self.rows).encode()) + for hashers in self.hashers: + for stream in self.STREAMS: + top.update(hashers[stream].digest()) + return top.hexdigest()[:16] + + +def logical_digest(batches, schema: pa.Schema) -> str: + d = LogicalDigest(schema) + for batch in batches: + d.update(batch) + return d.hexdigest() + + +def write_pinned(batches, schema: pa.Schema, dest: Path, combine: bool = True, **settings) -> str: + """Regroup a batch stream into fixed row groups and write it; return the digest of the file's bytes.""" + options = {"compression": "zstd", "compression_level": 3, "version": "2.6", "data_page_version": "1.0"} | settings + with pq.ParquetWriter(dest, schema, **options) as writer: + pending, rows = [], 0 + for batch in batches: + pending.append(batch) + rows += batch.num_rows + while rows >= ROW_GROUP: + table = pa.Table.from_batches(pending) + head, tail = table.slice(0, ROW_GROUP), table.slice(ROW_GROUP) + writer.write_table(head.combine_chunks() if combine else head, row_group_size=ROW_GROUP) + pending, rows = tail.to_batches(), tail.num_rows + if rows: + table = pa.Table.from_batches(pending) + writer.write_table(table.combine_chunks() if combine else table, row_group_size=ROW_GROUP) + return hashlib.sha256(dest.read_bytes()).hexdigest()[:16] + + +def make_table() -> pa.Table: + rng = np.random.default_rng(11) + floats = rng.random(N) + floats[rng.integers(0, N, 1000)] = np.nan + ints = rng.integers(0, 1_000_000, N) + return pa.table( + { + "id": pa.array(np.arange(N)), + "g": pa.array(ints), + "f": pa.array(floats, mask=rng.random(N) < 0.01), + "s": pa.array([f"k{v % 5000:05d}" for v in ints], mask=rng.random(N) < 0.01), + "b": pa.array(ints % 2 == 0), + "ts": pa.array(ints.astype("datetime64[s]")), + } + ) + + +def main() -> None: + table = make_table() + schema = table.schema + print(f"table: {N:,} rows, {table.nbytes / 1e6:.0f} MB of Arrow data, pyarrow {pa.__version__}\n") + with tempfile.TemporaryDirectory() as tmp: + tmp = Path(tmp) + + print("file bytes, the same rows arriving in different batch sizes:") + for combine in (True, False): + digests = { + size: write_pinned( + table.to_batches(max_chunksize=size), schema, tmp / f"c{int(combine)}_{size}.parquet", combine + ) + for size in (8_192, 100_000, N) + } + distinct = sorted(set(digests.values())) + print(f" combine_chunks per row group={combine!s:5}: {len(distinct)} distinct digest(s) {distinct}") + created_by = pq.ParquetFile(tmp / "c1_8192.parquet").metadata.created_by + print(f" footer created_by = {created_by!r} (moves with every writer upgrade)\n") + + print("file shape, the same rows:") + xorq_shape = tmp / "xorq_shape.parquet" + with pq.ParquetWriter(xorq_shape, schema, compression="snappy") as writer: + for batch in table.to_batches(max_chunksize=8_192): + writer.write_batch(batch) # one row group per batch, as xorq's ParquetStorage writes them + for label, path in (("zstd, 1,048,576-row groups", tmp / "c1_8192.parquet"), ("xorq's shape", xorq_shape)): + md = pq.ParquetFile(path).metadata + size = path.stat().st_size / 1e6 + groups, footer = md.num_row_groups, md.serialized_size / 1e3 + print(f" {label:27s}: {size:5.1f} MB, {groups:3d} row groups, {footer:6.1f} KB footer") + print() + + def of_file(path: Path, batch_size: int = 65_536) -> str: + pf = pq.ParquetFile(path) + return logical_digest(pf.iter_batches(batch_size=batch_size), pf.schema_arrow) + + print("logical content:") + by_batch = {size: logical_digest(table.to_batches(max_chunksize=size), schema) for size in (8_192, 100_000, N)} + distinct = len(set(by_batch.values())) + print(f" in-memory stream, batch sizes 8,192 / 100,000 / one table: {distinct} distinct digest(s)") + + zstd_path, snappy_path = tmp / "c1_8192.parquet", tmp / "snappy_small_groups.parquet" + pq.write_table(table, snappy_path, compression="snappy", row_group_size=8_192) + reference = of_file(zstd_path) + stored = pq.ParquetFile(zstd_path).schema_arrow + coerced = [f"{a.name}: {a.type} -> {b.type}" for a, b in zip(schema, stored) if a.type != b.type] + print(f" stream handed to the writer vs the file read back: same={by_batch[8_192] == reference}") + print(f" (the writer coerced {coerced})") + print(f" same file, read in 1,000-row batches: same={of_file(zstd_path, 1_000) == reference}") + print(f" same rows written snappy with 8,192-row groups: same={of_file(snappy_path) == reference}") + engine = xo.connect().read_parquet(str(zstd_path)).order_by("id").to_pyarrow_batches() + print(f" same file through DataFusion (read, order_by id): same={logical_digest(engine, stored) == reference}") + + def rewritten(changed: pa.Table) -> str: + path = tmp / "changed.parquet" + pq.write_table(changed, path, compression="zstd", row_group_size=ROW_GROUP) + return of_file(path) + + flipped = table.set_column(1, "g", pc.add(table["g"], pa.array(np.eye(1, N, 12345, dtype=np.int64)[0]))) + print(f" one value changed: differs={rewritten(flipped) != reference}") + f = table["f"].combine_chunks() + first_null = int(np.flatnonzero(f.is_null().to_numpy(zero_copy_only=False))[0]) + as_zero = pa.concat_arrays([f.slice(0, first_null), pa.array([0.0]), f.slice(first_null + 1)]) + print(f" one null replaced by 0.0: differs={rewritten(table.set_column(2, 'f', as_zero)) != reference}") + swapped = table.take(pa.array(np.r_[1, 0, np.arange(2, N)])) + print(f" first two rows swapped: differs={rewritten(swapped) != reference}\n") + + print("cost:") + t0 = time.perf_counter() + logical_digest(table.to_batches(max_chunksize=8_192), schema) + secs = time.perf_counter() - t0 + print(f" logical digest of an in-memory stream: {secs:.2f} s ({table.nbytes / 1e6 / secs:.0f} MB/s)") + t0 = time.perf_counter() + of_file(zstd_path) + secs = time.perf_counter() - t0 + print(f" logical digest of the file (read back + hash; the build and the verify path): {secs:.2f} s") + t0 = time.perf_counter() + hashlib.sha256(zstd_path.read_bytes()).hexdigest() + secs = time.perf_counter() - t0 + print(f" file-bytes digest to verify ({zstd_path.stat().st_size / 1e6:.0f} MB file): {secs:.2f} s") + + +if __name__ == "__main__": + main() diff --git a/scripts/spike_row_order_paging.py b/scripts/spike_row_order_paging.py new file mode 100644 index 0000000..115e28c --- /dev/null +++ b/scripts/spike_row_order_paging.py @@ -0,0 +1,159 @@ +"""ADR-008 evidence: paging by a baked ``__row_order`` column. + +``__row_order`` holds 0..N-1 in the physical row order of a file tallyman wrote. The script answers, on the +DEFAULT parallel connection with no engine settings: + +1. Is ``ORDER BY __row_order LIMIT n OFFSET k`` repeatable and correct, and what does it cost? +2. Does declaring the file's order to the engine (``WITH ORDER``) remove the sort? +3. Does a user sort on a column with ties page repeatably, without and with ``__row_order`` as the last key? +4. What does a range request (``__row_order >= k AND __row_order < k + n``) cost at depth, without and with a + parquet page index in the file? +5. Could tallyman alter a cheap recipe so the column survives a ``select`` that does not name it? A ``select`` is + an allow-list, so an unnamed column is dropped whether or not anyone can see it. ``carry_row_order`` below is + the rewrite. It works, and ADR-008 D3 rejects it in favour of a build error. +6. What do joins and a debugging copy do with the column's name? + +For reference it also times the first draft's approach, a bare LIMIT/OFFSET on a single-partition connection. + + uv run python scripts/spike_row_order_paging.py +""" + +from __future__ import annotations + +import hashlib +import statistics +import tempfile +import time +from pathlib import Path + +import numpy as np +import pandas as pd +import pyarrow as pa +import pyarrow.parquet as pq +import xorq.api as xo +import xorq.vendor.ibis.expr.operations as ops +from xorq.common.utils.graph_utils import replace_nodes + +ROW = "__row_order" +N = 3_000_000 +REQUESTS = 6 +OFFSETS = (0, 1_000_000, 2_900_000) + + +def carry_row_order(expr): + """Rewrite a row-preserving expression so ``__row_order`` reaches the output as the last column.""" + + def replacer(node, kwargs): + if kwargs: + node = node.__recreate__(kwargs) + if isinstance(node, ops.Project) and ROW in node.parent.schema and ROW not in node.values: + return ops.Project(node.parent, {**node.values, ROW: ops.Field(node.parent, ROW)}) + if isinstance(node, ops.DropColumns) and ROW in node.columns_to_drop: + keep = frozenset(c for c in node.columns_to_drop if c != ROW) + return ops.DropColumns(node.parent, keep) if keep else node.parent + return node + + out = replace_nodes(replacer, expr).to_expr() + if ROW in out.columns and out.columns[-1] != ROW: + out = out.select([c for c in out.columns if c != ROW] + [ROW]) + return out + + +def measure(expr, first_col: str = ROW) -> tuple[int, list[int], float]: + pages, firsts, secs = set(), set(), [] + for _ in range(REQUESTS): + t0 = time.perf_counter() + df = expr.execute() + secs.append(time.perf_counter() - t0) + pages.add(hashlib.md5(df.to_csv(index=False).encode()).hexdigest()) + firsts.add(int(df[first_col].iloc[0])) + return len(pages), sorted(firsts)[:3], statistics.median(secs) * 1000 + + +def report(label: str, expr, want: int) -> None: + n, firsts, ms = measure(expr) + print(f" {label}: {n} distinct pages / {REQUESTS}, first {ROW} {firsts} (want {want}), median {ms:.0f} ms") + + +def operators(con, expr) -> str: + plan = con.raw_sql("EXPLAIN " + xo.to_sql(expr)).to_pandas() + physical = plan[plan.iloc[:, 0] == "physical_plan"].iloc[0, 1] + return " <- ".join(line.strip().split(":")[0] for line in physical.splitlines() if line.strip()) + + +def main() -> None: + rng = np.random.default_rng(9) + cols = {"g": rng.integers(0, 200, N)} + cols |= {f"v{i}": rng.random(N) for i in range(12)} + cols[ROW] = np.arange(N) + table = pa.table(cols) + + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "snapshot.parquet" + pq.write_table(table, path, compression="zstd", row_group_size=1_048_576) + size = path.stat().st_size / 1e6 + print(f"file: {size:.0f} MB, {N:,} rows x {table.num_columns} columns, written in {ROW} order\n") + + con = xo.connect() + t = con.read_parquet(str(path)) + + print(f"1. default connection, ORDER BY {ROW} LIMIT 50 OFFSET k") + for k in OFFSETS: + report(f"offset {k:>9,}", t.order_by(ROW).limit(50, offset=k), k) + print(" operators:", operators(con, t.order_by(ROW).limit(50, offset=1_000_000))) + + print("\n2. the same, with the file's order declared to the engine") + ordered = xo.connect() + ordered.raw_sql( + f"CREATE EXTERNAL TABLE snapshot_ordered STORED AS PARQUET LOCATION '{path}' WITH ORDER ({ROW} ASC)" + ) + o = ordered.table("snapshot_ordered") + for k in OFFSETS: + report(f"offset {k:>9,}", o.order_by(ROW).limit(50, offset=k), k) + print(" operators:", operators(ordered, o.order_by(ROW).limit(50, offset=1_000_000))) + + print("\n3. a user sort on g (200 distinct values, so ties of about 15,000 rows), offset 100,000") + for label, keys in (("ORDER BY g", ["g"]), (f"ORDER BY g, {ROW}", ["g", ROW])): + n, _, ms = measure(t.order_by(keys).limit(50, offset=100_000)) + print(f" {label:26s}: {n} distinct pages / {REQUESTS}, median {ms:.0f} ms") + + print(f"\n4. a range request, {ROW} >= k AND {ROW} < k + 50") + indexed = Path(tmp) / "snapshot_page_index.parquet" + pq.write_table(table, indexed, compression="zstd", row_group_size=1_048_576, write_page_index=True) + for label, file in (("row-group statistics only", path), ("with a parquet page index", indexed)): + r = xo.connect().read_parquet(str(file)) + print(f" {label}:") + for k in OFFSETS: + report(f" rows from {k:>9,}", r.filter((r[ROW] >= k) & (r[ROW] < k + 50)).order_by(ROW), k) + + print("\n5. a cheap recipe that does not name the column") + recipe = t.filter(t.g < 100).select("g", "v0").mutate(z=t.v0 * 2) + print(f" as written: columns {list(recipe.columns)}") + carried = carry_row_order(recipe) + print(f" as altered: columns {list(carried.columns)}") + want = int(np.flatnonzero(cols["g"] < 100)[100_000]) + report("page at offset 100,000", carried.order_by(ROW).limit(50, offset=100_000), want) + dropped = carry_row_order(t.drop(ROW, "v11")) + last, kept = dropped.columns[-1], "v11" in dropped.columns + print(f" t.drop({ROW!r}, 'v11') as altered: last column {last!r}, v11 kept: {kept}") + + print("\n6. names") + mem = xo.connect() + a, b, c = ( + mem.create_table(name, pd.DataFrame({"k": [1, 2, 3], col: [10, 20, 30], ROW: [0, 1, 2]})) + for name, col in (("a", "x"), ("b", "y"), ("c", "z")) + ) + print(f" two-way join: {list(a.join(b, 'k').columns)}") + print(f" three-way join: {list(a.join(b, 'k').join(c, 'k').columns)}") + print(f" debugging copy: {list(a.mutate(__row_order_v1=a[ROW]).columns)}") + + print("\nreference: first draft's approach, bare LIMIT/OFFSET on a single-partition connection") + single = xo.connect() + single.raw_sql("SET datafusion.execution.target_partitions = 1") + s = single.read_parquet(str(path)) + for k in OFFSETS: + report(f"offset {k:>9,}", s.limit(50, offset=k), k) + + +if __name__ == "__main__": + main() diff --git a/scripts/spike_window_read_order.py b/scripts/spike_window_read_order.py new file mode 100644 index 0000000..2b4584d --- /dev/null +++ b/scripts/spike_window_read_order.py @@ -0,0 +1,148 @@ +"""ADR-008 evidence: which DataFusion settings make unsorted LIMIT/OFFSET windows repeatable and in file order? + +Writes parquet files above ``datafusion.optimizer.repartition_file_min_size`` whose rows are physically sorted by +``id``, then issues the same window request several times per configuration, for two plan shapes: + +* ``bare`` — ``read_parquet -> limit/offset`` (a worthy entry's snapshot read) +* ``shaped`` — ``read_parquet -> filter -> mutate -> limit/offset`` (a cheap entry's plan over a snapshot) + +For each it reports how many distinct pages came back, the first ids seen against the file-order answer, the +exchange operators in the physical plan (the governing variable: a window is in file order exactly when no +exchange operator sits between the scan and the limit), and the median latency. Two row-group sizes are used +because ``repartition_file_scans = false`` alone returns the file-order page for one and not for the other. + +It then times a full-scan aggregate under each configuration, which is the cost of applying an order-pinning +setting to a connection that also runs aggregates. + +A last section checks two ops that ``classify_build`` treats as cheap and that have more than a scan under them: +``union`` (``UnionExec`` emits one partition per input, so an exchange operator survives ``target_partitions = 1``) +and ``distinct`` (a hash aggregate). + + uv run python scripts/spike_window_read_order.py +""" + +from __future__ import annotations + +import hashlib +import statistics +import tempfile +import time +from pathlib import Path + +import numpy as np +import pyarrow as pa +import pyarrow.parquet as pq +import xorq.api as xo + +N = 3_000_000 +REQUESTS = 8 +EXCHANGES = ("RepartitionExec", "CoalescePartitionsExec", "SortPreservingMergeExec") +CONFIGS = { + "default": [], + "repartition_file_scans=false": ["SET datafusion.optimizer.repartition_file_scans = false"], + "rfs=false + round_robin=false": [ + "SET datafusion.optimizer.repartition_file_scans = false", + "SET datafusion.optimizer.enable_round_robin_repartition = false", + ], + "target_partitions=1": ["SET datafusion.execution.target_partitions = 1"], +} + + +def connect(stmts: list[str]): + con = xo.connect() + for stmt in stmts: + con.raw_sql(stmt) + return con + + +def shapes(t): + return {"bare": t, "shaped": t.filter(t.id % 3 != 0).mutate(z=t.id * 2)} + + +def file_order_first_id(shape: str, offset: int) -> int: + if shape == "bare": + return offset + # kept ids are 1, 2, 4, 5, 7, 8, ...: the k-th (0-based) is 3 * (k // 2) + 1 + (k % 2) + return 3 * (offset // 2) + 1 + (offset % 2) + + +def plan_summary(con, expr) -> str: + plan = con.raw_sql("EXPLAIN " + xo.to_sql(expr)).to_pandas() + physical = plan[plan.iloc[:, 0] == "physical_plan"].iloc[0, 1] + found = [name for name in EXCHANGES if name in physical] + groups = physical.split("file_groups={")[1].split(" group")[0] + return f"{groups} file group(s), exchanges: {', '.join(found) or 'none'}" + + +def cheap_ops_beyond_a_scan(tmp: Path) -> None: + rng = np.random.default_rng(1) + half = N // 2 + paths = [] + for i in range(2): + path = tmp / f"part{i}.parquet" + part = pa.table({"id": np.arange(half) + i * half, "g": rng.integers(0, 5_000, half), "v": rng.random(half)}) + pq.write_table(part, path, row_group_size=100_000) + paths.append(path) + print("\n=== cheap ops with more than a scan under them (two files, ids 0..N/2-1 and N/2..N-1)") + for label in ("default", "target_partitions=1"): + con = connect(CONFIGS[label]) + a, b = (con.read_parquet(str(path)) for path in paths) + for name, expr, offset in ( + ("union", a.union(b), 1_400_000), + ("distinct", a.select("g").distinct(), 2_000), + ): + window = expr.limit(50, offset=offset) + pages = {hashlib.md5(window.execute().to_csv(index=False).encode()).hexdigest() for _ in range(REQUESTS)} + physical = con.raw_sql("EXPLAIN " + xo.to_sql(window)).to_pandas() + physical = physical[physical.iloc[:, 0] == "physical_plan"].iloc[0, 1] + found = [op for op in (*EXCHANGES, "UnionExec") if op in physical] + print( + f"{label:30s} {name:8s} offset={offset:>9,}: {len(pages)} distinct pages / {REQUESTS} " + f"| operators: {', '.join(found) or 'none'}" + ) + + +def main() -> None: + rng = np.random.default_rng(3) + table = pa.table({"id": np.arange(N), "g": rng.integers(0, 50_000, N), "v": rng.random(N), "w": rng.random(N)}) + probe = xo.connect() + threshold = probe.raw_sql("SHOW datafusion.optimizer.repartition_file_min_size").to_pandas().iloc[0, 1] + partitions = probe.raw_sql("SHOW datafusion.execution.target_partitions").to_pandas().iloc[0, 1] + print(f"engine: repartition_file_min_size={threshold} target_partitions={partitions}") + + with tempfile.TemporaryDirectory() as tmp: + for row_group_size in (8_192, 100_000): + path = Path(tmp) / f"sorted_by_id_rg{row_group_size}.parquet" + pq.write_table(table, path, row_group_size=row_group_size) + print(f"\n=== {path.stat().st_size / 1e6:.1f} MB file, rows sorted by id, row groups of {row_group_size:,}") + for label, stmts in CONFIGS.items(): + con = connect(stmts) + t = con.read_parquet(str(path)) + for shape, expr in shapes(t).items(): + for offset in (0, 1_000_000): + window = expr.limit(50, offset=offset) + pages, firsts, secs = set(), set(), [] + for _ in range(REQUESTS): + t0 = time.perf_counter() + df = window.execute() + secs.append(time.perf_counter() - t0) + pages.add(hashlib.md5(df.to_csv(index=False).encode()).hexdigest()) + firsts.add(int(df["id"].iloc[0])) + want = file_order_first_id(shape, offset) + print( + f"{label:30s} {shape:6s} offset={offset:>9,}: {len(pages)} distinct pages / {REQUESTS}, " + f"first ids {sorted(firsts)[:3]} (file order: {want}), " + f"median {statistics.median(secs) * 1000:.0f} ms | {plan_summary(con, window)}" + ) + agg = t.group_by("g").agg(n=t.count(), s=t.v.sum()) + secs = [] + for _ in range(3): + t0 = time.perf_counter() + agg.execute() + secs.append(time.perf_counter() - t0) + print(f"{label:30s} full-scan aggregate (50k groups): median {statistics.median(secs) * 1000:.0f} ms") + cheap_ops_beyond_a_scan(Path(tmp)) + + +if __name__ == "__main__": + main() From 077e30810623bf5ea9e1bd62a9e3edab436da59c Mon Sep 17 00:00:00 2001 From: Paddy Mullen Date: Sun, 20 Sep 2026 10:31:43 -0400 Subject: [PATCH 002/111] docs(plans): fold design-session answers 7-10 into the cache redesign ADRs - ADR-009 D6 (new): create runs the query twice and compares digests, so a non-reproducible recipe is known from birth and its file is pinned; a cheap entry gets the same check and is materialized if it fails. Adds a Testing section. The entry's schema is read from the file. - ADR-007 D11 (new): one write at a time per project, by taking the existing project file lock around every write; replaces the per-entry lock. Two tallyman servers on one project is unsupported (#183). - ADR-007 D12 (new): files are deleted only by an explicit user action; the startup warm-up stops rewriting deleted files; no disk budget yet. - ADR-007 D6: tallyman stops remembering Buckaroo sessions and re-posts every time, with a session id derived from project, hash and view kind. - ADR-007 D9: the agreed order of work. Nothing starts before review. - ADR-007 open question 1: whether two kinds of entry survive. Unanswered. - Testing sections for ADR-007 and ADR-008, marking which tests are red on main today. Co-Authored-By: Claude Fable 5.1 --- .../ADR-007-tallyman-owned-materialization.md | 137 +++++++++++++++--- plans/ADR-008-row-order-of-reads.md | 29 +++- plans/ADR-009-digest-stability.md | 94 ++++++++++-- 3 files changed, 222 insertions(+), 38 deletions(-) diff --git a/plans/ADR-007-tallyman-owned-materialization.md b/plans/ADR-007-tallyman-owned-materialization.md index 404cd5d..5880e1f 100644 --- a/plans/ADR-007-tallyman-owned-materialization.md +++ b/plans/ADR-007-tallyman-owned-materialization.md @@ -1,8 +1,9 @@ # ADR: Tallyman owns result materialization (no xorq cache nodes in builds) - **Status:** Proposed (2026-09-18, revised 2026-09-20 in the grilling - session, which added the governing rule, decision D10, and the resolution - recorded under D5). Supersedes two decisions of + session, which added the governing rule, decisions D10 to D12, and the + resolution recorded under D5). Awaiting Paddy's review; nothing here is + implemented. Supersedes two decisions of `plans/ADR-006-read-path-loads-builds.md`: its D4 (chaining inlines the parent's cache node) and its D8 (the manifest records the snapshot key and reads assert it). Five other ADR-006 decisions keep their intent: D5 (the @@ -257,8 +258,8 @@ of every descendant would re-run the parent's expensive subgraph. record-batch stream, and writes the snapshot itself: - a unique temp name in the destination directory, then `os.replace`; -- under a per-hash cross-process `flock`, with the existence check repeated - inside the lock, so a second process waits and then finds the file; +- under the project's write lock (D11), with the existence check repeated + inside the lock, so a second writer waits and then finds the file; - it numbers the rows as it writes them, in a last column named `__row_order` (decision D2 of `plans/ADR-008-row-order-of-reads.md`); - it returns the digest of what it wrote. @@ -356,9 +357,16 @@ Buckaroo cannot be asked to display an entry whose query is still running. The same ordering now covers a snapshot that was deleted later: `load_session` waits for `ensure_materialized` before it posts anything. -Deleting a snapshot (the Cache page, a reset prune, a future budget eviction) -ends every live Buckaroo session whose plan reads it first, using the -session-eviction hook that ADR-006 decision D10 introduced. The next +The same rule covers the session itself. Buckaroo drops a session that has had +no browser attached for an hour, and tallyman's session map assumes a session +lives as long as the Buckaroo process, so an entry reopened after an idle hour +is handed an id Buckaroo no longer knows and the grid never loads. Tallyman +stops remembering sessions. The session id is derived from the project, the +content hash and the view kind (D10), and `load_session` posts `/load_expr` +every time, which Buckaroo already treats as a no-op for a session it has. + +Deleting a snapshot (D12) ends every live Buckaroo session whose plan reads it +first, using the session-eviction hook that ADR-006 decision D10 introduced. The next `/api/session` re-materializes and opens a new session. Without this, a page request against a deleted file would fail inside Buckaroo with the `At least one path is required` error above, which is exactly the kind of @@ -395,9 +403,19 @@ cache node has crept back in. Removing the cache node changes the hash of every worthy entry, and bare-read chaining changes every child's. This lands together with ADR-008's change to `tallyman_read_csv` and ADR-009's digest definition, behind a single corpus -rebuild, with the failing tests committed and seen red first. After the -rebuild, `~/.cache/xorq/result_cache` (14 GB) and the older leaks under -`~/.cache/xorq/parquet/` (2.6 GB) can be deleted by hand. +rebuild. After the rebuild, `~/.cache/xorq/result_cache` (14 GB) and the older +leaks under `~/.cache/xorq/parquet/` (2.6 GB) can be deleted by hand. + +Order of work, agreed 2026-09-20. Nothing starts until Paddy has reviewed all +three ADRs. + +1. One commit of failing tests covering everything the three ADRs change, plus + the audit's independent bugs, pushed and seen red on CI. +2. The redesign as one change, then the corpus rebuild. +3. The independent bugs that remain, which change no hash: the chart error + loop, eager notebook sessions, the staleness scan re-hashing every source + per entry, `tallyman pack` shipping the cache, the primary-key search that + never converges, and the over-broad stat-cache wipes. ### D10. Every diff is built as an entry before it is displayed @@ -440,6 +458,70 @@ putting an alias on an entry that already exists. page sooner, and it breaks the governing rule in both directions: Buckaroo runs tallyman's join, repeatedly, and tallyman never learns whether it succeeded. +### D11. One write at a time per project + +Every write takes the project's existing lock: a build, a materialization, a +promote and a recalc, as well as the checkpoint that takes it today. +`_project_lock` (`catalog_state.py:232-242`) is an OS file lock on +`.checkpoint.lock`, so it holds between the two processes of a normal tallyman, +the MCP server and the companion, which both build. It becomes re-entrant +within a process, since a promote builds and then checkpoints. + +This replaces the per-entry lock of this ADR's first draft and closes the audit +finding that two builds of one entry can end with the failing one deleting the +winner's directory (`build.py:447-468`, `618-627`). FastMCP runs tool calls on +a thread pool, so parallel tool calls did build at once. They now queue. + +Two tallyman servers pointed at one project is unsupported. Paddy: "you have +done something diabolical and deserve the results." The file lock would still +serialize their writes on one machine, and nothing else about them is safe, +because each holds in-process state the other never sees. +buckaroo-data/tallyman#183 tracks detecting that case and refusing to start. + +### D12. Files are deleted only by an explicit user action + +Nothing deletes a materialized file on its own, and nothing rewrites one except +opening an entry that needs it (D5). There is no disk budget yet. That is the +rewrite of `plans/ADR-003-result-cache-cost-rubric.md`, deferred. + +- The startup warm-up stops calling `cached_result_expr`. Today it rewrites + every snapshot that was deleted, which undoes the Cache page's delete button + and can block startup on one large file. +- An explicit delete skips an entry marked not reproducible (decision D6 of + `plans/ADR-009-digest-stability.md`), whose file cannot be recreated. +- Ephemeral diff entries (D10) follow the same rule and are not collected + automatically. + +## Testing + +Tests marked *red* fail on `main` today and belong in the failing-tests commit. +The others cannot fail before the code exists and ride with the change. + +- **Sentinel** (*red*, D8). With `XORQ_CACHE_DIR` pointing at an empty + directory, a build, a chained child build, a view, a delete and a reopen + leave that directory empty. +- **Concurrent builds** (*red*, D11). Two threads building the same entry both + return it, and the entry's directory is intact afterwards. +- **Forgotten session** (*red*, D6). After Buckaroo has dropped a session, + reopening the entry yields a grid that loads. +- **Warm-up leaves deleted files alone** (*red*, D12). A snapshot deleted before + startup is still absent after startup with no requests made. +- **Identity** (D3). A child's hash changes when its parent's snapshot path + changes and not when the file's bytes do, and a filter over an aggregate's + snapshot is classed cheap. +- **Files exist before anything runs** (D5). With an ancestor's snapshot + deleted, opening a descendant rewrites the ancestor first, verifies it, and + never executes a plan whose file is missing. +- **Cold equals warm** (D7). With `compute_cache/` removed, every snapshot an + entry needs is reproduced with its recorded digest. +- **Handoff** (D6). A worthy entry's grid is posted a view build of its + snapshot, and Buckaroo writes nothing under `compute_cache/result_cache/`. +- **Diffs** (D10). Opening a diff builds an entry under + `compute_cache/ephemeral_entries/` before Buckaroo is called, the checkpoint + does not zip it, and promoting it keeps the same content hash. +- **Explicit delete** (D12). Deleting a snapshot ends the sessions that read it, + and the next open rewrites it. + ## Consequences - **Retired:** the `.cache()` call and the source-read injection in @@ -484,15 +566,26 @@ tallyman's join, repeatedly, and tallyman never learns whether it succeeded. ## Open questions -1. **Deep cheap chains.** Nothing cuts the graph between cheap entries. Run - `tests/test_perf_chain_depth.py` against this design and decide whether a - node-count or `compile_seconds` threshold should make an otherwise cheap - entry worthy. -2. **An unfaithful parent's descendants.** Descendants of an entry whose heal - failed verification were computed from bytes that no longer exist. Nothing - flags them today, and nothing here does either. -3. **Eviction policy.** D6 says what eviction must do to live sessions. Which - snapshots to evict, and when, stays with the ADR-003 rewrite. -4. **Garbage collection of ephemeral entries.** D10 says where they live and - that they are deletable. When to delete them belongs with the eviction - policy of open question 3. +1. **Do two kinds of entry survive?** This is the largest open question of the + set and Paddy has not answered it. Materializing every entry when it is + created would remove the cheap and worthy classifier, the build error of + ADR-008 decision D3, the allow-list of ADR-008 decision D4, the view case in + D6, and open question 2 below, and it would give every entry a digest. An + entry built on another would always read the parent's file, so no graph + would be more than one entry deep, which is the parquet boundary Paddy wanted + in June. `plans/ADR-003-result-cache-cost-rubric.md` already proposes + admitting every result and evicting by budget. The cost is one file per + entry: one project measured 19 GB of cache for 779 MB of data while every + CSV revision wrote a file. His answer to question 4 ("materialize the + parquet if necessary") stands until he says otherwise. +2. **Deep cheap chains.** Nothing cuts the graph between cheap entries. Moot if + open question 1 is answered with one kind of entry. +3. **Where ordered copies of sources live.** ADR-008 decision D2 adds one per + parquet source. `csv_ordered/` is global today, is never collected, and is + not included by `tallyman pack`. If every entry is materialized, a root + entry's own file could serve as the ordered copy. + +Closed in the grilling session: an unfaithful parent's descendants (ADR-009 +decision D6 finds a non-reproducible entry when it is created and pins its +file, so it is never rewritten underneath its descendants); eviction policy +and the collection of ephemeral entries (D12). diff --git a/plans/ADR-008-row-order-of-reads.md b/plans/ADR-008-row-order-of-reads.md index a7e0eea..3ab6b25 100644 --- a/plans/ADR-008-row-order-of-reads.md +++ b/plans/ADR-008-row-order-of-reads.md @@ -1,7 +1,7 @@ # ADR: Row order of reads (every file carries `__row_order`, every page sorts by it) - **Status:** Proposed (2026-09-18, revised 2026-09-20 in the grilling - session). The first draft pinned row order with an engine setting. Paddy + session). Awaiting Paddy's review; nothing here is implemented. The first draft pinned row order with an engine setting. Paddy proposed baking a row-order column into every file tallyman writes and sorting every page by it. The measurements below favour that, so it is now the decision and the engine setting is the rejected alternative under D5. @@ -339,6 +339,33 @@ does not establish a split scan and is not what makes that test meaningful; its aggregate is), and `src/tallyman_xorq/source_cache.py:98`. `tests/test_tallyman_read_csv.py:159` already says about 10 MB. +## Testing + +Tests marked *red* fail on `main` today and belong in the failing-tests commit. +The others cannot fail before the code exists and ride with the change. + +- **Repeatable pages** (*red*, D5). Eight identical `/api/data` requests at a + deep offset into an entry whose file is larger than 10,485,760 bytes return + identical rows, for a worthy entry and for a cheap one. The sorted case is + Buckaroo's code and is tested there (buckaroo-data/buckaroo#974). +- **The column** (D2). Every file tallyman writes ends in `__row_order`, holding + `0..N-1` with no gaps, and a materialization replaces an inherited one. +- **Dropping it fails the build** (D3). A cheap recipe whose select list omits + the column raises a build error whose message contains the corrected select. + A worthy recipe that omits it builds. +- **Asking for an order renumbers** (D3). An `order_by` recipe's file is + numbered in the requested order, and assigning to the column is a build + error. +- **A debugging copy survives** (D6). `__row_order_v1` is still present, with + the parent's positions, after the child is materialized. +- **Classification** (D4). A union, a distinct, an unnest and an operation the + allow-list has never seen are all classed worthy. +- **Reserved** (D6). The primary-key search never returns `__row_order`, and a + diff has no `__row_order_v2` column. +- **CSV roots** (D7). A `tallyman_read_csv` entry has no Sort in its build, is + classed cheap, and has exactly one row-order column. +- **The hint** (D8). The `/load_expr` payload names `__row_order`. + ## Consequences - Pages are repeatable for unsorted and sorted requests, in tallyman and in diff --git a/plans/ADR-009-digest-stability.md b/plans/ADR-009-digest-stability.md index 5f9763c..0b34f87 100644 --- a/plans/ADR-009-digest-stability.md +++ b/plans/ADR-009-digest-stability.md @@ -1,8 +1,9 @@ # ADR: Digest stability (a heal is flagged only when the result changed) -- **Status:** Proposed (2026-09-18, revised 2026-09-20: D3 gains two format - requirements from `plans/ADR-008-row-order-of-reads.md`, and D1 lost its - speed gate). Amends +- **Status:** Proposed (2026-09-18, revised 2026-09-20 in the grilling + session: D3 gains two format requirements from + `plans/ADR-008-row-order-of-reads.md`, D1 lost its speed gate, and D6 is + new). Awaiting Paddy's review; nothing here is implemented. Amends `plans/ADR-004-result-digest-canonical-ordering.md` (Option A's "hash the snapshot bytes") and decision D5 of `plans/ADR-006-read-path-loads-builds.md` (the canonical sort), which said @@ -214,6 +215,11 @@ decoding a whole row group: 19 to 24 ms at any depth with the index against 79 to 90 ms without it, on a 287 MB file (`scripts/spike_row_order_paging.py` on the ADR-008 branch). +The entry's recorded schema (`schema.json`) is read from the written file and +not from the expression. The file is what every consumer reads, the writer +adds a column the graph does not have (`__row_order`), and parquet changes some +types on the way in: a `timestamp[s]` column comes back as `timestamp[ms]`. + Combining each row group keeps the file bytes reproducible as well. Nothing depends on that after D2, and it means two writes of the same entry can still be compared with `cmp` when debugging. Because of D2 these settings can change later without @@ -232,18 +238,81 @@ row. ### D5. It lands with the rebuild Every worthy entry's digest is recomputed by the corpus rebuild of ADR-007 -decision D9 (one change, one rebuild). -The manifest field keeps its name. Tests that cannot fail first ride with the -fix; the float-aggregate heal test and a batch-boundary digest test fail on -`main` and belong in the failing-tests commit. +decision D9 (one change, one rebuild). The manifest field keeps its name. + +### D6. Create runs the query twice and compares + +Decided by Paddy in the grilling session (2026-09-20): "call the same query +twice... put some tests around this." + +Today tallyman learns that an entry is not reproducible only when a deleted +file is rewritten and its digest differs. By then the original rows are gone, +and everything built on them disagrees with the new file without anyone +knowing. So the check moves to the moment the entry is created: + +- `materialize` runs the entry's query twice through the same writer, on the + same single-partition connection (D1), and compares the two content digests + (D2). Both runs go through the writer because D2's digest is defined on the + file as read back. The second file is then discarded. +- When they match, the entry is recorded as reproducible, with its digest. +- When they differ, the build still succeeds, because a recipe that calls + `sample()` is legitimate. The entry is recorded as not reproducible, its file + is pinned (never deleted by tallyman, ADR-007 decision D12), it is badged in + the UI, and the build result tells the author so, with the columns whose + digests differed. This is the state ADR-006 decision D12 (unfaithful entries + are pinned and badged) reaches after the damage is done. D6 reaches it first. +- A cheap entry gets the same check. Its two runs are streamed and digested + without writing a file. If they differ, the entry is materialized and pinned + like any other non-reproducible entry, because re-running it on every read + would show different values each time. A computed column that calls + `random()` is row-preserving, so it is classed cheap today, and nothing + detects it: ADR-006 decision D9 ("no cheap-entry digests") assumed a frozen + graph over pinned inputs always gives the same rows. + +The cost is a second execution of every create. It is accepted under the +priority recorded in ADR-007 ("a cohesive system that works reliably" first). + +A heal is still verified against the recorded digest, as now. After D6 a +mismatch there means something changed underneath a reproducible entry, which +D4 attributes. + +## Testing + +Tests marked *red* fail on `main` today and belong in the failing-tests commit. +The others cannot fail before the code exists and ride with the change. + +- **Float aggregate reproduces** (*red*). An entry with a float `SUM` and `AVG` + over a source large enough to run in parallel is created, its file deleted, + and the entry reopened, three times. Every rewrite must match the recorded + digest, and no `unfaithful_heal` record may be written. +- **Digest ignores batching and format.** The same rows delivered as 8,192-row + batches, 100,000-row batches and one table give one digest, and so do the + same rows written Snappy with small row groups. +- **Digest sees what it must.** One changed value, one null replaced by `0.0`, + and two swapped rows each change the digest. +- **Create detects a non-reproducible recipe.** A recipe whose UDF returns a + different value on every call builds successfully, is recorded as not + reproducible, names the offending column, and has a pinned file. +- **Create passes a reproducible recipe.** A deterministic recipe is recorded + as reproducible, and the query is observed to run exactly twice. +- **A non-reproducible cheap recipe is materialized.** A computed column that + calls `random()` ends up with a pinned file instead of re-running per read. +- **A pinned file survives an explicit delete.** The Cache page's delete skips + it and says why. +- **The schema comes from the file.** An entry with a `timestamp[s]` column + records `timestamp[ms]`, and every entry's recorded schema ends in + `__row_order`. ## Consequences - A float-aggregate entry heals to the same digest, and a pyarrow upgrade no longer flags anything. The loud responses of ADR-006 decisions D10 and D12 are left for the causes they were designed for. -- Materializations and heals are slower, by about 3x on aggregation at spike scale and up - to 7x in ADR-004's parking measurement. Reads are unaffected. +- Materializations and heals are slower, by about 3x on aggregation at spike + scale and up to 7x in ADR-004's parking measurement, and a create runs its + query twice (D6). Reads are unaffected. +- A recipe that is not reproducible is known to be so from the moment it is + created, and its file is never rewritten underneath the entries built on it. - Verify decodes the file instead of hashing its bytes. It runs on a heal and in `catalog_scan_staleness(verify_results=True)`, never on a read. - Snapshots are smaller, and their footers were 85 times smaller in the spike, @@ -258,11 +327,6 @@ fix; the float-aggregate heal test and a batch-boundary digest test fail on 1. **Nested types.** The spike covers fixed-width, boolean, string and binary columns. Lists, structs and maps need a recursive definition. -2. **Types parquet cannot store as given.** A `timestamp[s]` column comes back - as `timestamp[ms]`, so the snapshot's schema differs from the entry's - recorded schema. Either the writer refuses such a column, or the entry's - schema is recorded from the snapshot. `__row_order` pushes toward the second - answer, since the writer adds a column the entry's graph does not have. -3. **Engine upgrades.** Single-partition execution fixes the merge order within +2. **Engine upgrades.** Single-partition execution fixes the merge order within one engine version. Nothing guarantees float results across versions. D4 attributes that case and the remedy stays a rebuild. From 20d330de9f9a99a6eddacead6b27147d6c1b3aeb Mon Sep 17 00:00:00 2001 From: Paddy Mullen Date: Sun, 20 Sep 2026 11:44:17 -0400 Subject: [PATCH 003/111] docs(plans): fold the PR #184 review into the cache redesign ADRs An adversarial review of the three draft ADRs, and Paddy's decisions on its findings (2026-09-20). Still Proposed; nothing is implemented. - ADR-007 D10 (every diff is built as an entry) moved out to #188, so the set stays about the core structure of the cache. D10 is kept as a stub so that references to D11 and D12 stay valid. The live diff is recorded as the one known exception to the governing rule. - ADR-007 D5: the verify sweep is not a caller of ensure_materialized; it reads and never writes. The ordered-copy gap is recorded as accepted. - ADR-007 D6: the session id is derived from project and content hash (closes #172); the exact condition under which Buckaroo skips a repeat /load_expr; deleting a snapshot ends no session; an unfaithful heal posts force_reload, since evict_session worked by forgetting a record that D6 removes. - ADR-007 D11: three limits of the project lock (per-thread re-entrancy, recalc granularity left open, reads not covered, #118). #186 tracks the waiting it causes. - ADR-007 D12: a file is written only because something is about to read it. The warm-up sentence now states the precise condition. - ADR-007 D2: memoizing the read stops tables piling up, not footer reads. - ADR-008 D2, D7: CSVs do not already work this way. The intermediate is keyed by path and overwritten in place (#168), so D7 cannot land before that fix. CSVs go through source identity and the ordered copy is built from the content-addressed clone. - ADR-008: "repeatable pages" asserts the exact rows; a new test that an edited CSV forks the hash; memory at depth recorded under Consequences as a performance matter, deferred. - ADR-009 D1, D3: single-partition execution does not make an ungrouped float total a function of the rows alone; it depends on the parent file's row-group layout (#187). Row-group size and batch_size are pinned and the manifest records a snapshot format version. - ADR-009 D6: the cheap-entry half moved to #185; the limits of running twice are stated, and the #88 lint is mentioned. - ADR-009: four stale cross-references to ADR-008 fixed. - All three Testing sections: normal TDD. Every test goes in the failing-tests commit; a test of a missing function fails on import. New evidence scripts, each runnable from a clean temp dir: scripts/spike_csv_source_identity.py (ADR-008 D2, D7), scripts/spike_float_layout_digest.py (ADR-009 D1, D3), scripts/spike_deep_page_memory.py (ADR-008 Consequences). Co-Authored-By: Claude Fable 5.1 --- .../ADR-007-tallyman-owned-materialization.md | 257 ++++++++++++------ plans/ADR-008-row-order-of-reads.md | 106 ++++++-- plans/ADR-009-digest-stability.md | 147 +++++++--- scripts/spike_csv_source_identity.py | 77 ++++++ scripts/spike_deep_page_memory.py | 77 ++++++ scripts/spike_float_layout_digest.py | 83 ++++++ 6 files changed, 594 insertions(+), 153 deletions(-) create mode 100644 scripts/spike_csv_source_identity.py create mode 100644 scripts/spike_deep_page_memory.py create mode 100644 scripts/spike_float_layout_digest.py diff --git a/plans/ADR-007-tallyman-owned-materialization.md b/plans/ADR-007-tallyman-owned-materialization.md index 5880e1f..701518e 100644 --- a/plans/ADR-007-tallyman-owned-materialization.md +++ b/plans/ADR-007-tallyman-owned-materialization.md @@ -2,8 +2,10 @@ - **Status:** Proposed (2026-09-18, revised 2026-09-20 in the grilling session, which added the governing rule, decisions D10 to D12, and the - resolution recorded under D5). Awaiting Paddy's review; nothing here is - implemented. Supersedes two decisions of + resolution recorded under D5, and again the same day after a review of + PR #184: D10 moved out to #188, the verify sweep left D5's callers, D6's + session-ending clause was dropped, and D12's rule was restated). Awaiting + Paddy's review; nothing here is implemented. Supersedes two decisions of `plans/ADR-006-read-path-loads-builds.md`: its D4 (chaining inlines the parent's cache node) and its D8 (the manifest records the snapshot key and reads assert it). Five other ADR-006 decisions keep their intent: D5 (the @@ -19,6 +21,10 @@ buckaroo-data/buckaroo#972 (`/load_expr` has no `cache_dir`). The direction was set by Paddy the same day: "I want to depend on xorq as little as possible for caching." +- **Tickets:** #188 (diffs, moved out of this ADR), #186 (the waiting that + D11's lock causes), #185 (non-pure recipes), #183 (two servers on one + project), #168 (CSV source identity, which the shared rebuild of D9 needs), + #118 (concurrent reads on the shared backend, which D11 does not cover). - **Affected code:** `src/tallyman_xorq/source_cache.py` (`rewrite_for_build`), `src/tallyman_xorq/result_cache.py` (`_resolve_result_plan`, `cached_result_expr`, `entry_graph_expr`, `baked_snapshot_path`, @@ -154,6 +160,10 @@ computation, writes result files, or repairs tallyman's cache. Besides being simpler, this puts every failure of a computation in tallyman's process, where it can be logged and reported, and none inside a grid query in Buckaroo. +One exception is known and accepted for now. The live diff still hands +Buckaroo an unmaterialized join, because the decision that fixed it (D10) was +moved out of this set to #188. + ## Decisions ### D1. Builds carry no cache nodes @@ -201,9 +211,12 @@ snapshot comes from the manifest's `cache_worthy`, as it does now. the facade"). For a worthy entry it returns one bare read of `snapshot_path`, memoized per `(project, content_hash)` for the life of the process, so repeated reads stop -registering tables and stop re-opening the footer. A worthy entry whose file -exists is served without loading its build at all. For a cheap entry it returns -the loaded graph, as now. +piling up tables in the shared backend: one read has one table name. It does +not stop the footer being opened per query, because xorq registers a deferred +read's table again on every execute (`xorq/expr/api.py`, +`_transform_deferred_reads`). With the format of ADR-009 decision D3 that is a +2.7 KB read. A worthy entry whose file exists is served without loading its +build at all. For a cheap entry it returns the loaded graph, as now. Deleted: `manifest.snapshot_key`, `_assert_recorded_snapshot_key`, `_cached_node_path`, `rewrite_cache_dirs`, and the `cache_dir` argument. With @@ -292,8 +305,26 @@ badge (ADR-006 decision D12). Callers: the canonical read (`cached_result_expr`, on every call, where step 1 or a handful of `stat` calls is the whole cost, and which covers diff -composition), chaining at mint time (D3), `load_session` (D6), and the verify -sweep. +composition), chaining at mint time (D3), and `load_session` (D6). + +The verify sweep (`verify_sweep`, `staleness.py:118`, reached through +`catalog_scan_staleness(verify_results=True)`) is not a caller. It checks the +files that exist, reports each entry that recorded a digest as faithful, +unfaithful or absent, and writes nothing, which is what it does today. A sweep +that called this function would rewrite every deleted snapshot in the project, +which D12 forbids. Nothing is lost by leaving absent files alone: every file +this function writes is verified before it is served, so an absent file is +checked at the moment it next exists. A file that exists with the wrong digest +is reported through the same loud path and left in place, since deleting it is +the user's action (D12). + +One gap is known and accepted. Step 2 collects only reads under +`compute_cache/result_cache/`. A root entry also reads an ordered copy of its +source (decision D2 of `plans/ADR-008-row-order-of-reads.md`), which lives +elsewhere and which this function does not re-create. If one is missing, +Buckaroo fails with `At least one path is required`. Paddy, 2026-09-20: +Buckaroo erroring when tallyman has not provided a prerequisite is acceptable +for now, and follow-on work closes it. ADR-006 decision D4 rejected bare-read chaining because "builds stay non-self-contained and the pre-heal choreography stays load-bearing forever". @@ -361,16 +392,37 @@ The same rule covers the session itself. Buckaroo drops a session that has had no browser attached for an hour, and tallyman's session map assumes a session lives as long as the Buckaroo process, so an entry reopened after an idle hour is handed an id Buckaroo no longer knows and the grid never loads. Tallyman -stops remembering sessions. The session id is derived from the project, the -content hash and the view kind (D10), and `load_session` posts `/load_expr` -every time, which Buckaroo already treats as a no-op for a session it has. - -Deleting a snapshot (D12) ends every live Buckaroo session whose plan reads it -first, using the session-eviction hook that ADR-006 decision D10 introduced. The next -`/api/session` re-materializes and opens a new session. Without this, a page -request against a deleted file would fail inside Buckaroo with the -`At least one path is required` error above, which is exactly the kind of -failure the rule says belongs to tallyman. +stops remembering sessions. The session id is derived from the project and the +content hash, and `load_session` posts `/load_expr` with that id every time. +Buckaroo 0.15.6 skips the work when it already has a session with that id and +the same build directory, provided the post carries none of +`component_config`, `column_config_overrides`, `extra_grid_config`, `init_sd` +and `skip_stat_columns` (`buckaroo/server/handlers.py`, lines 429-454). +Tallyman sends none of them for an ordinary entry, so the repeat post is a +no-op there. A promoted diff entry sends `column_config_overrides` and so +reloads on every open, which #188 covers. Putting the project in the id also +closes #172 (one project's session served to another on a hash collision). + +Deleting a snapshot (D12) ends no session. A tab that already has the entry +open fails on its next query, inside Buckaroo, with the +`At least one path is required` error above. That is accepted: the user +deleted the file on purpose, and reopening the entry fixes it, because every +open goes through `ensure_materialized` before it posts. The session Buckaroo +still holds then works again. xorq registers the path afresh on every query, +and a faithful rewrite has the same content, so the session's stats are still +right. An earlier draft ended every session whose plan read the deleted file. +It was dropped in review: it needs an index from each snapshot to every entry +that reads it, and Buckaroo has no route that ends a session. + +An unfaithful heal is the one event that leaves a session wrong, since the +path now holds different rows and the session holds stats computed from the +old ones. ADR-006 decision D10 handles it today through `evict_session`, which +works by dropping tallyman's own record of the session so that the next load +mints a new one. With no record to drop, the companion's unfaithful-heal hook +wipes the entry's stat cache, as now, and posts `/load_expr` for the entry's +id with `force_reload: true`, which Buckaroo already accepts and which re-runs +its pipeline for that session. Sessions of entries built on the healed file +are stale too, as they are today; that is part of #185. This resembles what #104 removed: #102's viewer build over `/result.parquet`. #104's objection was two materialized copies per @@ -402,9 +454,11 @@ cache node has crept back in. Removing the cache node changes the hash of every worthy entry, and bare-read chaining changes every child's. This lands together with ADR-008's change to -`tallyman_read_csv` and ADR-009's digest definition, behind a single corpus -rebuild. After the rebuild, `~/.cache/xorq/result_cache` (14 GB) and the older -leaks under `~/.cache/xorq/parquet/` (2.6 GB) can be deleted by hand. +`tallyman_read_csv`, the fix for #168 that the change depends on (CSV sources +go through source identity, ADR-008 decision D2), and ADR-009's digest +definition, behind a single corpus rebuild. After the rebuild, +`~/.cache/xorq/result_cache` (14 GB) and the older leaks under +`~/.cache/xorq/parquet/` (2.6 GB) can be deleted by hand. Order of work, agreed 2026-09-20. Nothing starts until Paddy has reviewed all three ADRs. @@ -417,46 +471,28 @@ three ADRs. per entry, `tallyman pack` shipping the cache, the primary-key search that never converges, and the over-broad stat-cache wipes. -### D10. Every diff is built as an entry before it is displayed - -Live and promoted diffs already build the same expression -(`build_compare_expr`). The live path posts it to Buckaroo unmaterialized -(`_build_compare_expr`, `app.py:418-441`), so the outer join runs again for -every page, sort and stat query in the diff grid, and a failure of the join -surfaces inside Buckaroo. The promote path writes a recipe that calls -`build_diff_expr(a_hash, b_hash, keys)` and runs the normal build. - -Every diff now takes the promote path. When a diff view is opened, tallyman -builds the diff entry and waits for the build to finish. A diff contains a -join, so it is worthy and is materialized, and Buckaroo is then handed a view -build of the finished file (D6) with the diff's display configuration. The page -shows a "building diff" state while it waits. Promoting a diff is reduced to -putting an alias on an entry that already exists. - -- **An unnamed diff entry is ephemeral.** The checkpoint zips and commits every - complete directory under `entries/`, aliased or not (`zip_pending_entries`, - `catalog.py:160-180`), so an unnamed diff stored there would put every diff - ever viewed into the catalog's git history. Ephemeral entries live under - `compute_cache/ephemeral_entries//`, where everything is - already defined as deletable at any time, and they can be rebuilt from the - two hashes and the keys. Promote moves the directory into `entries/`, sets - the alias and checkpoints. The content hash is computed from the graph, so - the move does not change it, and the snapshot path stays the same. -- **Parents.** A worthy side is read from its snapshot. A cheap side is not - copied first: because the diff is itself materialized, each side is read - exactly once, while the diff is written. -- **Sessions.** The same entry can be opened as a diff, with the diff display - classes, or as a plain entry, so the Buckaroo session key includes the view - kind and is no longer the content hash alone. -- **Retired with this:** `_build_compare_expr` and its temp build directory, - the separate diff-session bookkeeping (`diff_session_is_loaded`, - `mark_diff_session_loaded`), and `diff_stat_cache/`, since a diff's Buckaroo - stats become its entry's own. The audit finding that every recalc wipes all - of `diff_stat_cache/` goes with it. - -*Rejected:* keep posting the join and materialize nothing. It shows a first -page sooner, and it breaks the governing rule in both directions: Buckaroo runs -tallyman's join, repeatedly, and tallyman never learns whether it succeeded. +### D10. Diffs: moved to a follow-on (#188) + +This decision said that every diff is built as an entry before it is +displayed, with an unnamed diff stored as an ephemeral entry under +`compute_cache/`. Paddy moved it out of this set on 2026-09-20, so that the set +stays about the core structure of the cache. Its text, and what the review +found about it, are in #188. The number is kept so that references to D11 and +D12 stay valid. + +What this set still does for diffs: + +- `build_compare_expr` drops `__row_order` from both sides before joining + (decision D6 of `plans/ADR-008-row-order-of-reads.md`). +- Both sides are read through `cached_result_expr`, so both files exist before + the join is composed (D5). A diff is the one consumer that needs two files + at once, which makes it the natural test of D5. +- A promoted diff is an ordinary worthy entry. It contains a join, so it is + materialized by D4 like any other. + +Until #188 lands the live diff works as it does today. It posts an +unmaterialized join to Buckaroo (`_build_compare_expr`, `app.py:418-441`), +which is the known exception recorded under the governing rule. ### D11. One write at a time per project @@ -472,6 +508,24 @@ finding that two builds of one entry can end with the failing one deleting the winner's directory (`build.py:447-468`, `618-627`). FastMCP runs tool calls on a thread pool, so parallel tool calls did build at once. They now queue. +Three limits, recorded here so that the implementation does not have to +discover them: + +- Re-entrant has to mean per thread. `_project_lock` takes `flock` on a fresh + file descriptor, so a nested acquire in one process blocks forever. The + companion also moves work between threads with `run_in_threadpool`, so the + lock cannot be held across an `await`. +- Whether a recalc takes the lock once per build or once for the whole walk is + left to the implementation. Either is correct. +- The lock covers writes only. Concurrent reads on the shared default backend + fail with `RuntimeError: Already borrowed`. That is #118, and this ADR does + not change it. + +The lock is blocking and has no timeout, and the work it now covers is long: a +materialization runs single-partition and twice (ADR-009 decisions D1 and D6). +A page request that needs a heal therefore waits behind any build in the other +process. Paddy, 2026-09-20: correct first. #186 tracks the waiting. + Two tallyman servers pointed at one project is unsupported. Paddy: "you have done something diabolical and deserve the results." The file lock would still serialize their writes on one machine, and nothing else about them is safe, @@ -480,31 +534,39 @@ buckaroo-data/tallyman#183 tracks detecting that case and refusing to start. ### D12. Files are deleted only by an explicit user action -Nothing deletes a materialized file on its own, and nothing rewrites one except -opening an entry that needs it (D5). There is no disk budget yet. That is the -rewrite of `plans/ADR-003-result-cache-cost-rubric.md`, deferred. - -- The startup warm-up stops calling `cached_result_expr`. Today it rewrites - every snapshot that was deleted, which undoes the Cache page's delete button - and can block startup on one large file. +Nothing deletes a materialized file on its own, and nothing writes one +speculatively. A file is written only because something is about to read it: a +read of the entry, or a build or a read of an entry whose plan reads its file +(a page request, a chart, chaining at mint time, a recalc). Those are D5's +callers. There is no disk budget yet. That is the rewrite of +`plans/ADR-003-result-cache-cost-rubric.md`, deferred. + +- The startup warm-up stops calling `cached_result_expr`. Today it heals + deleted snapshots until a 3 s budget is spent, and the budget is checked only + between entries (`app.py:807-824`). That undoes the Cache page's delete + button, and one large heal blocks startup for as long as it takes. +- The verify sweep reads and never writes (D5). - An explicit delete skips an entry marked not reproducible (decision D6 of - `plans/ADR-009-digest-stability.md`), whose file cannot be recreated. -- Ephemeral diff entries (D10) follow the same rule and are not collected - automatically. + `plans/ADR-009-digest-stability.md`), whose file cannot be recreated. The + skip protects the file from the Cache page only. `compute_cache/` as a whole + is still deletable by definition (D7), and where such a file should live is + part of #185. ## Testing -Tests marked *red* fail on `main` today and belong in the failing-tests commit. -The others cannot fail before the code exists and ride with the change. +Every test below goes in the failing-tests commit and is seen red on CI before +the change lands (D9, step 1). A test of a function that does not exist yet +fails on import, and that counts as red. An earlier draft let such tests ride +with the change. Paddy, 2026-09-20: do normal TDD. -- **Sentinel** (*red*, D8). With `XORQ_CACHE_DIR` pointing at an empty +- **Sentinel** (D8). With `XORQ_CACHE_DIR` pointing at an empty directory, a build, a chained child build, a view, a delete and a reopen leave that directory empty. -- **Concurrent builds** (*red*, D11). Two threads building the same entry both +- **Concurrent builds** (D11). Two threads building the same entry both return it, and the entry's directory is intact afterwards. -- **Forgotten session** (*red*, D6). After Buckaroo has dropped a session, +- **Forgotten session** (D6). After Buckaroo has dropped a session, reopening the entry yields a grid that loads. -- **Warm-up leaves deleted files alone** (*red*, D12). A snapshot deleted before +- **Warm-up leaves deleted files alone** (D12). A snapshot deleted before startup is still absent after startup with no requests made. - **Identity** (D3). A child's hash changes when its parent's snapshot path changes and not when the file's bytes do, and a filter over an aggregate's @@ -516,11 +578,15 @@ The others cannot fail before the code exists and ride with the change. entry needs is reproduced with its recorded digest. - **Handoff** (D6). A worthy entry's grid is posted a view build of its snapshot, and Buckaroo writes nothing under `compute_cache/result_cache/`. -- **Diffs** (D10). Opening a diff builds an entry under - `compute_cache/ephemeral_entries/` before Buckaroo is called, the checkpoint - does not zip it, and promoting it keeps the same content hash. -- **Explicit delete** (D12). Deleting a snapshot ends the sessions that read it, - and the next open rewrites it. +- **Two files at once** (D5, D10). With both sides' snapshots deleted, + composing a diff rewrites and verifies both before the join is built, and the + diff carries no row-order column from either side. +- **Explicit delete** (D12). After a snapshot is deleted, the next open + rewrites it and verifies it before Buckaroo is called. +- **The sweep writes nothing** (D5, D12). With a snapshot deleted, a verify + sweep reports the entry as absent, and the file is still absent afterwards. +- **Forced reload** (D6). After an unfaithful heal, Buckaroo receives a + `/load_expr` for that entry's session id with `force_reload` set. ## Consequences @@ -529,7 +595,9 @@ The others cannot fail before the code exists and ride with the change. `manifest.snapshot_key` and `_assert_recorded_snapshot_key`; the `baked` / `recompute` plan split keyed on a `CachedNode`; the thread-only `_heal_lock` and the `(FileNotFoundError, ValueError)` retry around xorq's shared temp - file; `entry_graph_expr` as a separate function. + file; `entry_graph_expr` as a separate function; tallyman's record of + Buckaroo sessions (`_sessions`, `~/.tallyman/buckaroo_sessions.json`) and + `evict_session`, which worked by dropping an entry of it (D6). - **ADR-006:** its D4 (inlined chaining) and D8 (snapshot-key tripwire) are superseded. Its D2's "snapshot path derived from the loaded expression" becomes `snapshot_path`. Its D3 (rebind composition onto the default backend) @@ -562,7 +630,10 @@ The others cannot fail before the code exists and ride with the change. become unnecessary. Not tested. - **Cost accepted:** a worthy parent's snapshot must exist before a child can be built. A build also no longer repairs its own ancestors when something - outside tallyman executes it, and under the governing rule nothing does. + outside tallyman executes it, and under the governing rule nothing does. A + tab open on an entry whose file the user deletes errors until the entry is + reopened (D6). A missing ordered copy of a source surfaces as a Buckaroo + error (D5). ## Open questions @@ -582,10 +653,20 @@ The others cannot fail before the code exists and ride with the change. open question 1 is answered with one kind of entry. 3. **Where ordered copies of sources live.** ADR-008 decision D2 adds one per parquet source. `csv_ordered/` is global today, is never collected, and is - not included by `tallyman pack`. If every entry is materialized, a root - entry's own file could serve as the ordered copy. + not included by `tallyman pack`. Its path is also outside the project root, + so `make_portable_inplace` does not rewrite it and a CSV entry's build is + not portable. The fix for #168 proposes keeping a CSV's ordered copy under + the project, next to the content-addressed clone it is built from, which + would settle this for CSVs. If every entry is materialized, a root entry's + own file could serve as the ordered copy. + +Closed in the grilling session: eviction policy (D12). -Closed in the grilling session: an unfaithful parent's descendants (ADR-009 +Reopened in review and moved out of this set: an unfaithful parent's +descendants. The grilling session closed it on the grounds that ADR-009 decision D6 finds a non-reproducible entry when it is created and pins its -file, so it is never rewritten underneath its descendants); eviction policy -and the collection of ephemeral entries (D12). +file. The pin holds against the Cache page only, the file lives under +`compute_cache/`, which D7 defines as deletable, and an entry built on a +non-reproducible parent is itself recorded as reproducible, because both of +its runs read the same parent file. #185 has it. The collection of ephemeral +diff entries went to #188 with D10. diff --git a/plans/ADR-008-row-order-of-reads.md b/plans/ADR-008-row-order-of-reads.md index 3ab6b25..6cdae2d 100644 --- a/plans/ADR-008-row-order-of-reads.md +++ b/plans/ADR-008-row-order-of-reads.md @@ -1,7 +1,9 @@ # ADR: Row order of reads (every file carries `__row_order`, every page sorts by it) - **Status:** Proposed (2026-09-18, revised 2026-09-20 in the grilling - session). Awaiting Paddy's review; nothing here is implemented. The first draft pinned row order with an engine setting. Paddy + session, and again the same day after a review of PR #184, which made the + fix for #168 a precondition of D2 and D7). Awaiting Paddy's review; nothing + here is implemented. The first draft pinned row order with an engine setting. Paddy proposed baking a row-order column into every file tallyman writes and sorting every page by it. The measurements below favour that, so it is now the decision and the engine setting is the rejected alternative under D5. @@ -10,10 +12,15 @@ threshold quoted in `plans/ADR-006-read-path-loads-builds.md` decision D5 (the canonical sort) and three other places. - **Context:** the 2026-09-18 cache audit (tallyman @ `a748ea6`, buckaroo - 0.15.4, xorq 0.3.26, xorq-datafusion 0.2.7). No ticket filed yet. + 0.15.4, xorq 0.3.26, xorq-datafusion 0.2.7). +- **Tickets:** #168 (CSV sources bypass source identity; D2 and D7 depend on + its fix), buckaroo-data/buckaroo#974 (Buckaroo's half of D5, see D8), #188 + (diffs, moved out of ADR-007). - **Affected code:** `src/tallyman_xorq/source_cache.py` (`rewrite_for_build`, `_tie_break_order`), `src/tallyman_xorq/io.py` (`read_project_file`, - `tallyman_read_csv`, `io.py:627`), `src/tallyman_xorq/result_cache.py` + `tallyman_read_csv`, `io.py:627`, and for #168 `_ordered_csv_key` and + `_ordered_csv_parquet`), `src/tallyman_xorq/source_identity.py` (the three + steps a CSV now goes through), `src/tallyman_xorq/result_cache.py` (`_EXPENSIVE_OPS`, `classify_build`), `src/tallyman_companion/app.py` (`api_data`, `app.py:930`, and the chart data it feeds), `src/tallyman_xorq/primary_key.py` (candidate selection, @@ -26,10 +33,12 @@ (what a snapshot is, and the corpus rebuild this shares), `plans/ADR-009-digest-stability.md` (the file format, which D5 adds a requirement to). -- **Evidence:** `scripts/spike_row_order_paging.py` (the decisions) and +- **Evidence:** `scripts/spike_row_order_paging.py` (the decisions), `scripts/spike_window_read_order.py` (the problem, and the rejected - engine-setting approach). All figures are from those scripts on a 14-core - machine. + engine-setting approach), `scripts/spike_csv_source_identity.py` (D2 and D7: + what a CSV edit does to a content hash) and + `scripts/spike_deep_page_memory.py` (the memory figures under Consequences). + All figures are from those scripts on a 14-core machine. ## Terms @@ -135,12 +144,25 @@ Two writers produce it: positions in its own file. - **Ingest.** A source file enters tallyman through an ordered copy: polars scans it, `with_row_index` numbers the rows in file order, and the copy is - written with the column last. CSVs already work this way (the intermediate - parquet under `csv_ordered/`). Parquet sources gain the same step, keyed by - the source's digest, beside the content-addressed clone that stays the + written with the column last. The copy is keyed by the source's content and + written once. Both kinds of source go through source identity first + (`si.digest_for`, then `si.ensure_cas_path`, then `si.note_source`, the three + steps `read_project_file` performs for parquet today, `io.py:78-84`), and the + ordered copy is built from the content-addressed clone, which stays the immutable input. A source that already has a `__row_order` column has it overwritten, which is the right outcome for a file tallyman exported. + CSVs have the ordered-copy step today and not the keying. The first draft of + this decision said they already worked this way, which was wrong. The + intermediate under `csv_ordered/` is keyed by `md5(absolute path | schema | + reader options)` (`io.py:515-527`), it is overwritten in place when the CSV's + mtime changes (`io.py:558-582`), and `tallyman_read_csv` never touches source + identity. That is #168, and D7 cannot land before its fix. Building the copy + from the clone is enough: the key function already hashes the path, and the + clone's path carries the digest, which is the device of + `plans/ADR-002-source-identity-content-hash.md`. The clone never changes, so + the in-place overwrite becomes dead code and is deleted. + The canonical sort's tie-break (`_tie_break_order`) puts an inherited `__row_order` where `original_row_order` is today: after the author's own `order_by` keys and before the remaining columns. A worthy entry that keeps its @@ -290,8 +312,10 @@ semantics, in any engine and any process. every table would win the search for any table without a string or id key. Row positions shift between versions, so a diff keyed on it would be meaningless. -- `build_compare_expr` drops it from both sides before joining. The diff is an - entry (ADR-007 decision D10) and gets its own when it is materialized. +- `build_compare_expr` drops it from both sides before joining. A promoted + diff is a worthy entry, since it contains a join, and gets its own when it is + materialized. The live diff grid has none until #188 lands (diffs built as + entries, moved out of ADR-007), so its paging stays as it is today. ### D7. `tallyman_read_csv` loses its trailing `order_by`, and its column becomes `__row_order` @@ -301,13 +325,26 @@ and written last, so a CSV root has one row-order column and not two with identical values. The root entry becomes a cheap read of the intermediate, and D5 gives file-order pages with no Sort and no second copy. +This depends on the fix for #168 (D2). Today the trailing Sort is the only +thing that keeps a CSV root's rows fixed: it makes the root worthy, so a baked +snapshot freezes them, and a heal that read a changed intermediate would be +flagged. Without the Sort and without the fix, editing a CSV and re-running the +same recipe gives the same content hash, so no new version is created, and the +first version's frozen build returns the edited rows (`[10, 999, 30, 40]` where +it was built from `[10, 20, 30]`). With the ordered copy keyed on the clone, the +edit gives a new hash and the first version keeps its rows +(`scripts/spike_csv_source_identity.py`). The hash does not move today either, +with the Sort in place, so an edited CSV under an unchanged recipe produces no +new version. The fix for #168 corrects that as well. + What INV-2 provided, and what replaces it: | INV-2 gave | Replacement | | --- | --- | | A canonical display order | D5. INV-2 did not deliver this above 10 MB. | | A parquet boundary for chained children | A cheap root's graph is one read node. | -| A `result_digest` on the root, so a re-parse that produced different rows would be caught | Lost as it stands: cheap entries record no digest (ADR-006 decision D9, "no cheap-entry digests"). See open question 5. | +| Fixed rows under the root's hash, through its baked snapshot | The content-keyed ordered copy of D2, written once. Requires the fix for #168. | +| A `result_digest` on the root, so a re-parse that produced different rows would be caught | Lost as it stands: cheap entries record no digest (ADR-006 decision D9, "no cheap-entry digests"). It matters only when an ordered copy is deleted and re-created. See open question 5. | Every hash in every CSV lineage changes, so this rides the corpus rebuild of ADR-007 decision D9 ("one change, one rebuild"). @@ -341,13 +378,21 @@ its aggregate is), and `src/tallyman_xorq/source_cache.py:98`. ## Testing -Tests marked *red* fail on `main` today and belong in the failing-tests commit. -The others cannot fail before the code exists and ride with the change. - -- **Repeatable pages** (*red*, D5). Eight identical `/api/data` requests at a - deep offset into an entry whose file is larger than 10,485,760 bytes return - identical rows, for a worthy entry and for a cheap one. The sorted case is - Buckaroo's code and is tested there (buckaroo-data/buckaroo#974). +Every test below goes in the failing-tests commit and is seen red on CI before +the change lands (ADR-007 decision D9, the order of work). A test of a function +that does not exist yet fails on import, and that counts as red. Paddy, +2026-09-20: do normal TDD. + +- **Repeatable pages** (D5). Eight identical `/api/data` requests at a deep + offset into an entry whose file is larger than 10,485,760 bytes each return + exactly the rows at positions `offset` to `offset + limit - 1` in + `__row_order` order, for a worthy entry and for a cheap one. Asserting only + that the eight agree is not enough: one rejected setting returned the same + wrong page 8 times out of 8 (D5). The sorted case is Buckaroo's code and is + tested there (buckaroo-data/buckaroo#974). +- **An edited CSV forks the hash** (D2, #168). Editing a CSV and re-running the + same recipe gives a new content hash, and the earlier entry still returns the + rows it was built from. - **The column** (D2). Every file tallyman writes ends in `__row_order`, holding `0..N-1` with no gaps, and a materialization replaces an inherited one. - **Dropping it fails the build** (D3). A cheap recipe whose select list omits @@ -379,10 +424,18 @@ The others cannot fail before the code exists and ride with the change. applies to them. - Unions, distincts and unnests are materialized, which is the price of having a defined row order. -- Each parquet source costs one ordered copy, about the size of the source. +- Each source costs one ordered copy, about the size of the source. A CSV + source also gains a content-addressed clone of the CSV, which is + copy-on-write where the filesystem offers it. - An unsorted page costs a sort of one column unless the file's order is declared (100 to 374 ms against 25 to 242 ms in the spike). A range request costs about 20 ms at any depth. +- A sorted page also holds more in memory the deeper it is. On the spike's + file (336 MB as Arrow) peak process memory was 626 MB at offset 0, 1,142 MB + at 1,000,000 and 1,176 MB at 2,900,000, against 300 MB for a bare limit and + 248 MB for a range request (`scripts/spike_deep_page_memory.py`). The corpus + holds a 3.68 GB snapshot. Paddy, 2026-09-20: a performance matter, taken up + after correctness. The range request is the path with bounded memory. - The grid stays unstable until Buckaroo's half (buckaroo-data/buckaroo#974) lands: above 10 MB when unsorted, and at any size when sorted by a column with ties. @@ -403,7 +456,12 @@ The others cannot fail before the code exists and ride with the change. 4. **Deep offsets on a cheap entry.** Positions in a filtered view have gaps, so it pages with `OFFSET`, whose cost grows with depth. 5. **A digest for an ordered copy.** An ordered copy is the record of a parse - and is re-created from the clone if deleted. Recording its digest in the root - entry's manifest, and verifying it on re-creation, would restore what D7 - gives up. It wants ADR-009's digest definition, and it touches where the - copies live: `csv_ordered` is global, is never collected, and is not packed. + and is re-created from the clone if deleted. With the copy keyed by content + (D2) a given path always holds the same parse, so what is left unverified is + a re-creation, for example after a polars upgrade that parses differently. + Recording the copy's digest in the root entry's manifest, and verifying it + on re-creation, would cover that. It wants ADR-009's digest definition, and + it touches where the copies live: `csv_ordered` is global, is never + collected, and is not packed (ADR-007 open question 3). Nothing re-creates a + missing copy automatically yet, which ADR-007 decision D5 (one entry point + makes files exist) records as an accepted gap. diff --git a/plans/ADR-009-digest-stability.md b/plans/ADR-009-digest-stability.md index 0b34f87..1003329 100644 --- a/plans/ADR-009-digest-stability.md +++ b/plans/ADR-009-digest-stability.md @@ -3,7 +3,9 @@ - **Status:** Proposed (2026-09-18, revised 2026-09-20 in the grilling session: D3 gains two format requirements from `plans/ADR-008-row-order-of-reads.md`, D1 lost its speed gate, and D6 is - new). Awaiting Paddy's review; nothing here is implemented. Amends + new; and again the same day after a review of PR #184: D1 and D3 now say + what single-partition execution leaves undetermined, and D6's cheap-entry + half moved to #185). Awaiting Paddy's review; nothing here is implemented. Amends `plans/ADR-004-result-digest-canonical-ordering.md` (Option A's "hash the snapshot bytes") and decision D5 of `plans/ADR-006-read-path-loads-builds.md` (the canonical sort), which said @@ -13,7 +15,9 @@ always means this ADR's own decision. Another ADR's decision is always written with its ADR number and a few words saying what it decides. - **Context:** the 2026-09-18 cache audit (tallyman @ `a748ea6`, xorq 0.3.26, - xorq-datafusion 0.2.7, pyarrow 21.0.0). No ticket filed yet. + xorq-datafusion 0.2.7, pyarrow 21.0.0). +- **Tickets:** #187 (a float total depends on the parent file's layout; D1 and + D3), #185 (non-pure recipes, which D6 only partly covers). - **Affected code:** `src/tallyman_xorq/result_cache.py` (`snapshot_file_digest`, `verify_result_faithful`, `_verify_self_heal`), `src/tallyman_xorq/build.py` (the execute-once step), @@ -23,11 +27,14 @@ introduces (one writer for snapshots, used by the build and by every heal). D2 and D3 assume that writer. If ADR-007 were rejected, D1 would stand as written and D2 would need restating against xorq's writer. -- **Related ADRs:** `plans/ADR-008-row-order-of-reads.md` (uses the same - single-partition setting for a different job, and has an open question this +- **Related ADRs:** `plans/ADR-008-row-order-of-reads.md` (its first draft used + the same single-partition setting for page reads, and it now rejects that; + its open question 5, a digest for an ordered copy of a source, is one this digest would answer). - **Evidence:** `scripts/spike_float_aggregate_digest.py`, - `scripts/spike_logical_digest.py`. + `scripts/spike_logical_digest.py`, and + `scripts/spike_float_layout_digest.py` (D1 and D3: what the layout of the + parent file does to a float total). ## Terms @@ -106,11 +113,12 @@ rebuilt. `materialize` (ADR-007 decision D4, the one writer of snapshots) executes on a connection configured with -`SET datafusion.execution.target_partitions = 1`. The build and every heal use -it, so both run one plan with one merge order, on any machine. It is a -different connection object from ADR-008's window connection, with the same -setting, because a long materialization must not share a context with page -reads. +`SET datafusion.execution.target_partitions = 1` and an explicit +`datafusion.execution.batch_size`. The build and every heal use it, so both run +one plan with one merge order, on any machine, given input files with the same +layout (see "What this does not fix" below). It is a separate connection from +the default backend that serves page reads, because a long materialization +must not share a context with them. Cost: about 3x on the spike's aggregate. ADR-004 measured a 3.1M-group aggregate at 0.5 s parallel against 3.4 to 3.9 s single-partition, and a full @@ -130,6 +138,28 @@ single-partition only for a plan with a floating-point reduction or window, decided once at build from the expression and recorded in the manifest so that a heal never re-derives it. +**What this does not fix.** A single stream fixes the order in which partial +results are merged. It does not make a float total a function of the rows +alone. With one partition and the same rows in the same order, an ungrouped +float `SUM` or `AVG` took three different bit patterns across four copies of +one file that differed only in row-group size (1,048,576, 777,777, 100,000 and +8,192 rows), each stable run to run (`scripts/spike_float_layout_digest.py`). +The variable is association, meaning where the running total is cut into +sub-sums. The likely mechanism, not confirmed in DataFusion's source, is that +the ungrouped accumulator sums each record batch as a block while batch +boundaries follow row-group boundaries. Sorting the input first changes +nothing, because DataFusion removes the sort: the plan is +`AggregateExec <- DataSourceExec` with or without it. A grouped aggregate, +`GROUP BY` a constant key, and the window form `SUM(v) OVER ()` were all +independent of the layout, so an ibis percent-of-total is not affected. Only a +true ungrouped reduction is exposed. + +So the layout of every file an entry reads is part of what makes its digest +reproducible. Tallyman writes all of those files (snapshots through D3, and +sources through the ordered copies of ADR-008 decision D2), which is why +pinning the layout is enough. D3 pins it. #187 tracks the rest: confirming the +mechanism, other reductions, and results across machines. + *Rejected:* round floats before hashing. A value next to a rounding boundary flips under one unit of noise in the last place, and among millions of values some are. @@ -212,8 +242,7 @@ ADR-008 adds two requirements. The writer numbers the rows in a last column named `__row_order` (ADR-008 decision D2). And it writes a parquet page index, which is what lets a page be fetched as a range of `__row_order` values without decoding a whole row group: 19 to 24 ms at any depth with the index against 79 -to 90 ms without it, on a 287 MB file (`scripts/spike_row_order_paging.py` on -the ADR-008 branch). +to 90 ms without it, on a 287 MB file (`scripts/spike_row_order_paging.py`). The entry's recorded schema (`schema.json`) is read from the written file and not from the expression. The file is what every consumer reads, the writer @@ -222,18 +251,28 @@ types on the way in: a `timestamp[s]` column comes back as `timestamp[ms]`. Combining each row group keeps the file bytes reproducible as well. Nothing depends on that after D2, and it means two writes of the same entry can still -be compared with `cmp` when debugging. Because of D2 these settings can change later without -touching any digest. +be compared with `cmp` when debugging. + +Because of D2 the codec, the compression level and the statistics can change +later without touching this entry's digest. The row-group size cannot. It +decides the batch boundaries that an entry built on this file sees, and an +ungrouped float total depends on them (D1). An earlier draft said every +setting here was free to change. The row-group size and the materialization +connection's `batch_size` are therefore part of the reproducibility contract: +the manifest records a snapshot format version that stands for both, and +changing either is a corpus rebuild. ### D4. A mismatch record names its likely cause With D1 and D2 in place the causes left are the recipe's own nondeterminism (`sample()`, `now()`, an impure UDF), source drift under `off` identity mode, and an engine upgrade that changed results. The manifest records the xorq, -xorq-datafusion and pyarrow versions at build, and the `unfaithful_heal` record -carries them alongside the versions at heal. When they differ, the message says -so instead of blaming the recipe. The contract's attribution table gains that -row. +xorq-datafusion and pyarrow versions at build, together with the snapshot +format version (D3), and the `unfaithful_heal` record carries them alongside +the values at heal. When they differ, the message says so instead of blaming +the recipe. The contract's attribution table gains that row. One cause has no +attribution yet: a parent that is not reproducible was rewritten, so every +entry built on it heals to different rows. That belongs to #185. ### D5. It lands with the rebuild @@ -245,10 +284,13 @@ decision D9 (one change, one rebuild). The manifest field keeps its name. Decided by Paddy in the grilling session (2026-09-20): "call the same query twice... put some tests around this." -Today tallyman learns that an entry is not reproducible only when a deleted -file is rewritten and its digest differs. By then the original rows are gone, -and everything built on them disagrees with the new file without anyone -knowing. So the check moves to the moment the entry is created: +Today the build lint from #88 (`_nondeterminism_warnings`, `build.py`) warns +when a recipe uses one of five known non-pure operations, and nothing records +its verdict. Beyond that warning, tallyman learns that an entry is not +reproducible only when a deleted file is rewritten and its digest differs. By +then the original rows are gone, and everything built on them disagrees with +the new file without anyone knowing. So a check moves to the moment a +materialized entry is created: - `materialize` runs the entry's query twice through the same writer, on the same single-partition connection (D1), and compares the two content digests @@ -261,16 +303,27 @@ knowing. So the check moves to the moment the entry is created: the UI, and the build result tells the author so, with the columns whose digests differed. This is the state ADR-006 decision D12 (unfaithful entries are pinned and badged) reaches after the damage is done. D6 reaches it first. -- A cheap entry gets the same check. Its two runs are streamed and digested - without writing a file. If they differ, the entry is materialized and pinned - like any other non-reproducible entry, because re-running it on every read - would show different values each time. A computed column that calls - `random()` is row-preserving, so it is classed cheap today, and nothing - detects it: ADR-006 decision D9 ("no cheap-entry digests") assumed a frozen - graph over pinned inputs always gives the same rows. - -The cost is a second execution of every create. It is accepted under the -priority recorded in ADR-007 ("a cohesive system that works reliably" first). +- A cheap entry is not checked here. An earlier draft ran it twice as well, and + materialized and pinned it when the runs differed. That made a third kind of + entry, a graph the classifier calls cheap with a file the manifest says + exists, and whether an entry has a file stopped being a function of its + graph, which decisions D2, D3 and D5 of + `plans/ADR-007-tallyman-owned-materialization.md` all rely on. Paddy moved + it to #185 on 2026-09-20. Until that is settled a cheap entry that calls + `random()` behaves as it does today: the #88 lint warns, and the entry + re-runs on every read. + +The cost is a second execution of every create of a materialized entry. It is +accepted under the priority recorded in ADR-007 ("a cohesive system that works +reliably" first). + +Running twice has two limits, both part of #185. It cannot see `today()`, +since both runs agree; the #88 lint does. And it cannot see what an entry +inherits: an entry built on a non-reproducible parent reads the same parent +file in both runs and is recorded as reproducible, which holds only while that +file survives. The pin protects the file from the Cache page's delete and from +nothing else, since `compute_cache/` is deletable by definition (ADR-007 +decision D7, the cold state is an empty `compute_cache`). A heal is still verified against the recorded digest, as now. After D6 a mismatch there means something changed underneath a reproducible entry, which @@ -278,10 +331,12 @@ D4 attributes. ## Testing -Tests marked *red* fail on `main` today and belong in the failing-tests commit. -The others cannot fail before the code exists and ride with the change. +Every test below goes in the failing-tests commit and is seen red on CI before +the change lands (ADR-007 decision D9, the order of work). A test of a function +that does not exist yet fails on import, and that counts as red. Paddy, +2026-09-20: do normal TDD. -- **Float aggregate reproduces** (*red*). An entry with a float `SUM` and `AVG` +- **Float aggregate reproduces** (D1). An entry with a float `SUM` and `AVG` over a source large enough to run in parallel is created, its file deleted, and the entry reopened, three times. Every rewrite must match the recorded digest, and no `unfaithful_heal` record may be written. @@ -295,8 +350,9 @@ The others cannot fail before the code exists and ride with the change. reproducible, names the offending column, and has a pinned file. - **Create passes a reproducible recipe.** A deterministic recipe is recorded as reproducible, and the query is observed to run exactly twice. -- **A non-reproducible cheap recipe is materialized.** A computed column that - calls `random()` ends up with a pinned file instead of re-running per read. +- **The layout is pinned** (D1, D3). Every snapshot has row groups of 1,048,576 + rows, the materialization connection reports the pinned `batch_size`, and + the manifest records the snapshot format version. - **A pinned file survives an explicit delete.** The Cache page's delete skips it and says why. - **The schema comes from the file.** An entry with a `timestamp[s]` column @@ -311,8 +367,10 @@ The others cannot fail before the code exists and ride with the change. - Materializations and heals are slower, by about 3x on aggregation at spike scale and up to 7x in ADR-004's parking measurement, and a create runs its query twice (D6). Reads are unaffected. -- A recipe that is not reproducible is known to be so from the moment it is - created, and its file is never rewritten underneath the entries built on it. +- A materialized entry whose recipe is not reproducible is known to be so from + the moment it is created, and the Cache page's delete leaves its file alone. + What that does not yet cover (a cheap entry, an entry that inherits the + problem from its parent, a file lost with `compute_cache/`) is #185. - Verify decodes the file instead of hashing its bytes. It runs on a heal and in `catalog_scan_staleness(verify_results=True)`, never on a read. - Snapshots are smaller, and their footers were 85 times smaller in the spike, @@ -320,8 +378,10 @@ The others cannot fail before the code exists and ride with the change. - `docs/system-contract.md` changes in "Result digest" and in the manifest table ("SHA-256 of the baked result snapshot"), and its verification table gains the engine-change row. -- The same function can digest the CSV intermediate (ADR-008, open - question 1). +- The same function can digest an ordered copy of a source (ADR-008, open + question 5). +- The row-group size and the materialization connection's `batch_size` are + frozen for the life of the corpus, and changing either is a rebuild (D3). ## Open questions @@ -330,3 +390,8 @@ The others cannot fail before the code exists and ride with the change. 2. **Engine upgrades.** Single-partition execution fixes the merge order within one engine version. Nothing guarantees float results across versions. D4 attributes that case and the remedy stays a rebuild. +3. **What else decides a float's low bits.** D1 and D3 pin the two variables + found so far, the merge order and the batch boundaries. #187 tracks the + mechanism, other reductions (variance, correlation, window frames), a filter + between the scan and the aggregate, and results across machines and CPU + architectures. diff --git a/scripts/spike_csv_source_identity.py b/scripts/spike_csv_source_identity.py new file mode 100644 index 0000000..423b65f --- /dev/null +++ b/scripts/spike_csv_source_identity.py @@ -0,0 +1,77 @@ +"""ADR-008 evidence (D2, D7) and #168: is a CSV root's ordered intermediate fixed under a content hash? + +``tallyman_read_csv`` reads a CSV through an ordered parquet intermediate. That file's key is +``md5(absolute path | schema | reader options)``, not content, and it is overwritten in place when the CSV's mtime +changes. The script edits a CSV, re-runs the identical recipe, and reports the build hash before and after, and what +the first version's frozen build returns afterwards, for three shapes of the root expression: + +1. today: a read of the intermediate with a trailing ``order_by("original_row_order")``; +2. ADR-008 D7 without the fix for #168: a plain read of the same intermediate (no sort, so no snapshot, no digest); +3. the fix for #168: digest the CSV, clone it to a content-named path, and key the intermediate on the clone. + +Raw xorq builds, no tallyman build machinery, so case 1 reports the hash only: in a real build the root is worthy +because of its Sort, and its baked snapshot goes on serving the old rows. An unchanged hash also means +``build_and_persist`` returns the existing entry, so no new version is created at all. + + uv run python scripts/spike_csv_source_identity.py +""" + +from __future__ import annotations + +import os +import tempfile +import time +from pathlib import Path + +HOME = Path(tempfile.mkdtemp(prefix="spike_csv_identity_")) +os.environ["TALLYMAN_HOME"] = str(HOME) # the ordered intermediates live under it; set before tallyman is imported + +import xorq.api as xo # noqa: E402 +from xorq.ibis_yaml.compiler import build_expr, load_expr # noqa: E402 + +from tallyman_xorq import source_identity as si # noqa: E402 +from tallyman_xorq.io import _ordered_csv_parquet, tallyman_read_csv # noqa: E402 + +BUILDS = HOME / "builds" +CAS = HOME / "cas" +ORIGINAL = "id,amount\n1,10\n2,20\n3,30\n" +EDITED = "id,amount\n1,10\n2,999\n3,30\n4,40\n" + + +def today(csv: Path): + return tallyman_read_csv(str(csv)) + + +def plain_read(csv: Path): + return xo.deferred_read_parquet(str(_ordered_csv_parquet(str(csv), None, {}))) + + +def keyed_on_clone(csv: Path): + clone = CAS / f"{si._digest_file(csv)}{csv.suffix}" + if not clone.exists(): + clone.write_bytes(csv.read_bytes()) # tallyman's ensure_cas_path makes a copy-on-write clone here + return xo.deferred_read_parquet(str(_ordered_csv_parquet(str(clone), None, {}))) + + +def main() -> None: + CAS.mkdir() + csv = HOME / "sales.csv" + cases = ( + ("1. today (trailing order_by kept)", today, False), + ("2. ADR-008 D7 without the #168 fix", plain_read, True), + ("3. keyed on a content-named clone", keyed_on_clone, True), + ) + for label, root, reread in cases: + csv.write_text(ORIGINAL) + v1 = Path(build_expr(root(csv), builds_dir=BUILDS)) + time.sleep(0.05) # so the edit changes the CSV's mtime + csv.write_text(EDITED) + v2 = Path(build_expr(root(csv), builds_dir=BUILDS)) + print(f"{label}: hash before the edit {v1.name}, after {v2.name}, same hash: {v1.name == v2.name}") + if reread: + rows = load_expr(v1).execute().amount.tolist() + print(f" V1's frozen build, re-read after the edit: {rows} (built from {[10, 20, 30]})") + + +if __name__ == "__main__": + main() diff --git a/scripts/spike_deep_page_memory.py b/scripts/spike_deep_page_memory.py new file mode 100644 index 0000000..edb3d6e --- /dev/null +++ b/scripts/spike_deep_page_memory.py @@ -0,0 +1,77 @@ +"""ADR-008 evidence (Consequences): what a page request holds in memory at depth. + +ADR-008 D5 serves every page as ``ORDER BY __row_order LIMIT n OFFSET k``. Its cost table measures latency. This +script measures peak process memory for the same requests, on the same file shape as ``spike_row_order_paging.py``, +and for a range request (``__row_order >= k AND __row_order < k + n``), which ADR-008 notes as a later optimization. + +Each case runs in a fresh process, so its peak is its own; the first case imports xorq and does nothing, as the floor. + + uv run python scripts/spike_deep_page_memory.py +""" + +from __future__ import annotations + +import resource +import subprocess +import sys +import tempfile +from pathlib import Path + +ROW = "__row_order" +N = 3_000_000 +CASES = ( + "import only", + "bare limit 50", + "sorted 0", + "sorted 1000000", + "sorted 2900000", + "range 2900000", + "chart 100000", +) + + +def run_case(path: str, case: str) -> None: + import xorq.api as xo + + kind, _, arg = case.partition(" ") + if kind != "import": + t = xo.connect().read_parquet(path) + if kind == "bare": + t.limit(50).execute() + elif kind == "sorted": + t.order_by(ROW).limit(50, offset=int(arg)).execute() + elif kind == "range": + k = int(arg) + t.filter((t[ROW] >= k) & (t[ROW] < k + 50)).order_by(ROW).execute() + elif kind == "chart": + t.order_by(ROW).limit(int(arg)).execute() # the chart pull through /api/data + peak = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss + peak_mb = peak / 1e6 if sys.platform == "darwin" else peak / 1e3 # bytes on macOS, kilobytes on Linux + print(f" {case:16s} peak process memory {peak_mb:7.0f} MB") + + +def main() -> None: + import numpy as np + import pyarrow as pa + import pyarrow.parquet as pq + + rng = np.random.default_rng(9) + cols = {"g": rng.integers(0, 200, N)} + cols |= {f"v{i}": rng.random(N) for i in range(12)} + cols[ROW] = np.arange(N) + table = pa.table(cols) + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "snapshot.parquet" + pq.write_table(table, path, compression="zstd", row_group_size=1_048_576, write_page_index=True) + on_disk, as_arrow = path.stat().st_size / 1e6, table.nbytes / 1e6 + print(f"file: {on_disk:.0f} MB on disk, {as_arrow:.0f} MB as Arrow, {N:,} rows x {table.num_columns} columns") + for case in CASES: + out = subprocess.run([sys.executable, __file__, str(path), case], capture_output=True, text=True) + print(out.stdout.rstrip() or out.stderr.strip()[-300:]) + + +if __name__ == "__main__": + if len(sys.argv) == 3: + run_case(sys.argv[1], sys.argv[2]) + else: + main() diff --git a/scripts/spike_float_layout_digest.py b/scripts/spike_float_layout_digest.py new file mode 100644 index 0000000..21b0bc3 --- /dev/null +++ b/scripts/spike_float_layout_digest.py @@ -0,0 +1,83 @@ +"""ADR-009 evidence (D1, D3) and #187: what single-partition execution leaves undetermined in a float aggregate. + +``target_partitions = 1`` removes the run-to-run drift of a float aggregate. This script asks whether the result is +then a function of the rows alone. It writes the same rows, in the same order, as four parquet files that differ only +in row-group size, and bit-compares several query shapes across them on a single-partition connection: + +* ``A`` an ungrouped ``SUM`` / ``AVG``; +* ``B`` the same over a subquery sorted by row position, to test whether ordering the addends helps; +* ``C`` ``GROUP BY`` a literal key; +* ``D`` ``GROUP BY`` a single-valued key the optimizer cannot fold (``ro - ro``); +* ``E`` the window form ``SUM(v) OVER ()``, which is what an ibis percent-of-total compiles to. + +Every shape is stable run to run on one file. Shape A, and B with it, differs between files. The physical plan +printed for B shows why sorting changes nothing: DataFusion removes the sort, since ``SUM`` needs no ordered input. +The variable is association (where the running total is cut into sub-sums), not the order of the addends. + + uv run python scripts/spike_float_layout_digest.py +""" + +from __future__ import annotations + +import struct +import tempfile +from pathlib import Path + +import numpy as np +import pyarrow as pa +import pyarrow.parquet as pq +import xorq.api as xo + +N = 3_000_000 +ROW_GROUP_SIZES = (1_048_576, 777_777, 100_000, 8_192) +SHAPES = { + "A ungrouped": "SELECT SUM(v) s, AVG(v) m FROM {t}", + "B sorted first": "SELECT SUM(v) s, AVG(v) m FROM (SELECT * FROM {t} ORDER BY ro)", + "C group by literal": "SELECT SUM(v) s, AVG(v) m FROM (SELECT v, 1 AS k FROM {t}) GROUP BY k", + "D group by ro - ro": "SELECT SUM(v) s, AVG(v) m FROM (SELECT v, ro - ro AS k FROM {t}) GROUP BY k", + "E window SUM() OVER ()": "SELECT MIN(tot) s, MAX(tot) m FROM (SELECT SUM(v) OVER () AS tot FROM {t})", +} + + +def bits(value) -> str: + return struct.pack(" str: + plan = con.raw_sql("EXPLAIN " + sql).to_pandas() + physical = plan[plan.iloc[:, 0] == "physical_plan"].iloc[0, 1] + return " <- ".join(line.strip().split(":")[0] for line in physical.splitlines() if line.strip()) + + +def main() -> None: + rng = np.random.default_rng(5) + table = pa.table({"ro": np.arange(N), "v": rng.random(N) * 1e6}) + results: dict[str, dict[int, tuple[str, str]]] = {shape: {} for shape in SHAPES} + plans: dict[str, str] = {} + with tempfile.TemporaryDirectory() as tmp: + for size in ROW_GROUP_SIZES: + path = Path(tmp) / f"rows_rg{size}.parquet" + pq.write_table(table, path, row_group_size=size, compression="zstd") + con = xo.connect() + con.raw_sql("SET datafusion.execution.target_partitions = 1") + name = f"t_{size}" + con.read_parquet(str(path), table_name=name) + for shape, sql in SHAPES.items(): + runs = set() + for _ in range(2): + row = con.raw_sql(sql.format(t=name)).to_pandas().iloc[0] + runs.add((bits(row.s), bits(row.m))) + assert len(runs) == 1, f"{shape} is not stable run to run on row groups of {size:,}" + results[shape][size] = runs.pop() + plans.setdefault(shape, operators(con, sql.format(t=name))) + + print(f"{N:,} rows, the same order in every file; target_partitions = 1; row-group sizes {ROW_GROUP_SIZES}\n") + for shape, by_size in results.items(): + distinct = len(set(by_size.values())) + low_bytes = [by_size[size][0][:4] for size in ROW_GROUP_SIZES] + print(f"{shape:24s} distinct results across the four files: {distinct} SUM, lowest two bytes: {low_bytes}") + print(f"{'':24s} plan: {plans[shape]}") + + +if __name__ == "__main__": + main() From f777d067ba3693bc16a68acb2814f71f249906da Mon Sep 17 00:00:00 2001 From: Paddy Mullen Date: Sun, 20 Sep 2026 23:29:35 -0400 Subject: [PATCH 004/111] docs(plans): fold the second PR #184 review into the cache redesign ADRs Keep two kinds of entry (ADR-007 open question 1). ADR-007 gains D13 (a file is cache only if ensure_materialized can re-create it) and D14 (a reset leaves compute_cache alone), closes the ordered-copy gap in D5, and adds the klass hot-reload fix to D6. ADR-008 rewrites D4 as a three-part test for cheap, corrects D6's three-way join claim, and adds D10 (natural order on every order_by), D11 (hoist a non-final sort) and D12 (a raw parquet read is a build error). ADR-009 says how a loaded build gets onto the single-partition connection and extends the format version to ordered copies. Adds seven evidence scripts under scripts/. Items not yet confirmed by Paddy are marked as such in the ADR text. Co-Authored-By: Claude Sonnet 5 --- .../ADR-007-tallyman-owned-materialization.md | 303 ++++++++++++++--- plans/ADR-008-row-order-of-reads.md | 320 ++++++++++++++++-- plans/ADR-009-digest-stability.md | 74 +++- scripts/spike_cheap_classifier.py | 108 ++++++ scripts/spike_ordered_copy_layout.py | 100 ++++++ scripts/spike_reset_roundtrip.py | 93 +++++ scripts/spike_row_order_joins.py | 93 +++++ .../spike_single_partition_loaded_build.py | 98 ++++++ scripts/spike_sort_grafting.py | 139 ++++++++ scripts/spike_stream_order.py | 82 +++++ 10 files changed, 1324 insertions(+), 86 deletions(-) create mode 100644 scripts/spike_cheap_classifier.py create mode 100644 scripts/spike_ordered_copy_layout.py create mode 100644 scripts/spike_reset_roundtrip.py create mode 100644 scripts/spike_row_order_joins.py create mode 100644 scripts/spike_single_partition_loaded_build.py create mode 100644 scripts/spike_sort_grafting.py create mode 100644 scripts/spike_stream_order.py diff --git a/plans/ADR-007-tallyman-owned-materialization.md b/plans/ADR-007-tallyman-owned-materialization.md index 701518e..464a312 100644 --- a/plans/ADR-007-tallyman-owned-materialization.md +++ b/plans/ADR-007-tallyman-owned-materialization.md @@ -4,7 +4,10 @@ session, which added the governing rule, decisions D10 to D12, and the resolution recorded under D5, and again the same day after a review of PR #184: D10 moved out to #188, the verify sweep left D5's callers, D6's - session-ending clause was dropped, and D12's rule was restated). Awaiting + session-ending clause was dropped, and D12's rule was restated, and a third + time that day after a second review of PR #184: two kinds of entry were + confirmed (open question 1), D5 lost its accepted gap, D6 gained the klass + reload, and D13 and D14 are new). Awaiting Paddy's review; nothing here is implemented. Supersedes two decisions of `plans/ADR-006-read-path-loads-builds.md`: its D4 (chaining inlines the parent's cache node) and its D8 (the manifest records the snapshot key and @@ -24,7 +27,11 @@ - **Tickets:** #188 (diffs, moved out of this ADR), #186 (the waiting that D11's lock causes), #185 (non-pure recipes), #183 (two servers on one project), #168 (CSV source identity, which the shared rebuild of D9 needs), - #118 (concurrent reads on the shared backend, which D11 does not cover). + #118 (concurrent reads on the shared backend, which D11 does not cover), + #77 (an empty grid on a clone of the project at another path, which D2 + closes), #76 (closed: how bare-read chaining failed the last time it was + tried, which D5 answers), #22 (the checkpoint's cost grows with the cache, + which D14 removes). - **Affected code:** `src/tallyman_xorq/source_cache.py` (`rewrite_for_build`), `src/tallyman_xorq/result_cache.py` (`_resolve_result_plan`, `cached_result_expr`, `entry_graph_expr`, `baked_snapshot_path`, @@ -33,7 +40,11 @@ `src/tallyman_xorq/io.py` (`tracked_expr_from_alias`, `pinned_expr_from_alias`), `src/tallyman_xorq/portable.py` (`rewrite_cache_dirs`), `src/tallyman_core/manifest.py` (`snapshot_key`), - `src/tallyman_companion/buckaroo_lifecycle.py` (`load_session`), + `src/tallyman_companion/buckaroo_lifecycle.py` (`load_session`, + `reload_project_sessions`), `src/tallyman_core/catalog_state.py` + (`reset_to`, `prune_compute_cache`, `restore_from_bullpen`, + `capture_tallyman_state`, `_gc_cas_clones`), + `src/tallyman_xorq/source_identity.py` (`gc_cas`, `recon_cas_path`), `docs/system-contract.md`. - **Related ADRs:** `plans/ADR-002-source-identity-content-hash.md` (content in the path; D3 reuses the device), `plans/ADR-003-result-cache-cost-rubric.md` @@ -41,8 +52,11 @@ `plans/ADR-008-row-order-of-reads.md` and `plans/ADR-009-digest-stability.md` (the other two hash- or digest-changing decisions that share this ADR's corpus rebuild). -- **Evidence:** `scripts/spike_bare_read_chaining.py` (results under D3), and - the audit measurements quoted in the Problem section. +- **Evidence:** `scripts/spike_bare_read_chaining.py` (results under D3), + `scripts/spike_reset_roundtrip.py` (D13 and D14: a reset back and forward, + played through on today's code), `scripts/spike_stream_order.py` (open + question 1, the alternative that was not taken), and the audit measurements + quoted in the Problem section. ## Terms @@ -72,6 +86,17 @@ **promoted diff** is one that has been saved as a catalog entry. - **Checkpoint:** tallyman's step that zips new entries and makes one git commit in the catalog repository. +- **Clone:** the copy of a source file under `data/.cas/`, named by the digest + of the source's bytes, which a build reads in place of the live file + (`plans/ADR-002-source-identity-content-hash.md`). +- **Ordered copy:** the parquet copy of a source that carries `__row_order`, + which a root entry reads (decision D2 of + `plans/ADR-008-row-order-of-reads.md`). +- **Reset:** `reset_to`, which returns the catalog to an earlier step. The + **bullpen** is the directory a reset moves retired files into, so that a + later reset forward can bring them back. +- **Klass:** a summary stat, post-processing or display class written for the + project, which Buckaroo loads into each grid. ## Problem @@ -271,8 +296,12 @@ of every descendant would re-run the parent's expensive subgraph. record-batch stream, and writes the snapshot itself: - a unique temp name in the destination directory, then `os.replace`; -- under the project's write lock (D11), with the existence check repeated - inside the lock, so a second writer waits and then finds the file; +- under the project's write lock (D11). A heal repeats the existence check + inside the lock, so a second reader waits and then finds the file. A create + does not look: it always runs the query and replaces whatever is at the + path. That keeps an entry that is added again after a reset honest (D14), + and decision D6 of `plans/ADR-009-digest-stability.md` (create runs the + query twice) needs it; - it numbers the rows as it writes them, in a last column named `__row_order` (decision D2 of `plans/ADR-008-row-order-of-reads.md`); - it returns the digest of what it wrote. @@ -291,10 +320,12 @@ and keeps items 3 and 4. entry's plan reads is on disk before anything executes: 1. If the entry is worthy and its snapshot exists, return. No build is loaded. -2. Otherwise load the entry's build and collect the snapshot paths its `Read` - nodes point at (any read under `compute_cache/result_cache/`). The list is - kept with the loaded plan in the existing LRU. -3. For each of those that is missing, recurse on the hash in its file name. +2. Otherwise load the entry's build and collect every file its `Read` nodes + point at. The list is kept with the loaded plan in the existing LRU. +3. Re-create each of those that is missing, by the rule for its class (D13): a + snapshot by recursing on the hash in its file name, an ordered copy of a + source from its clone, a clone from the live source while the bytes still + match. 4. If the entry is worthy, `materialize` it. Every file it writes is verified against the manifest's `result_digest` before @@ -318,13 +349,19 @@ checked at the moment it next exists. A file that exists with the wrong digest is reported through the same loud path and left in place, since deleting it is the user's action (D12). -One gap is known and accepted. Step 2 collects only reads under -`compute_cache/result_cache/`. A root entry also reads an ordered copy of its -source (decision D2 of `plans/ADR-008-row-order-of-reads.md`), which lives -elsewhere and which this function does not re-create. If one is missing, -Buckaroo fails with `At least one path is required`. Paddy, 2026-09-20: -Buckaroo erroring when tallyman has not provided a prerequisite is acceptable -for now, and follow-on work closes it. +An earlier draft stopped at snapshots. Step 2 collected only reads under +`compute_cache/result_cache/`, and a missing ordered copy of a source +(decision D2 of `plans/ADR-008-row-order-of-reads.md`) surfaced inside +Buckaroo as `At least one path is required`. Paddy accepted that on +2026-09-20 as the case of someone deleting a directory by hand. The second +review played a reset through and found the same state reachable by an +ordinary reset back and forward, on today's code +(`scripts/spike_reset_roundtrip.py`, results under D13). After decision D7 of +`plans/ADR-008-row-order-of-reads.md` (a CSV root becomes a cheap read) every +page of a CSV-rooted cheap entry reads that copy, so the gap is closed here +and not left to follow-on work. When nothing can re-create a file, because the +clone is gone and the live source has changed, the error names the source +file. ADR-006 decision D4 rejected bare-read chaining because "builds stay non-self-contained and the pre-heal choreography stays load-bearing forever". @@ -336,6 +373,14 @@ through ordinary cache mechanics on first query." Under the governing rule Buckaroo never does that, so the property has no user and nothing is given up. This was question 1 of the grilling session, resolved 2026-09-20. +Bare-read chaining has failed here once before. #76 (closed) is a child +frozen on its parent's snapshot path and read after a reset had pruned that +snapshot, or on a clone of the project that never had it: the read died with +`At least one path is required` and nothing recomputed the parent. ADR-006 +decision D4 (chaining inlines the parent's cache node) was the answer then. +This function is the answer now, and the +"files exist before anything runs" test below is #76's reproduction. + The old pre-heal was also weaker than this function. It was a `cached_result_expr` call with a discarded result at one call site, and ancestors were healed only because reads then re-executed recipes. Here the @@ -403,6 +448,17 @@ no-op there. A promoted diff entry sends `column_config_overrides` and so reloads on every open, which #188 covers. Putting the project in the id also closes #172 (one project's session served to another on a hash collision). +One caller still needs to know which grids are open. When a klass is added or +changed, `reload_project_sessions` (`buckaroo_lifecycle.py:386`, four call +sites in `app.py`) posts `/reload_expr/` to every session of the +project, and it finds them in the record this decision deletes. Buckaroo +0.15.6 has no route that lists sessions. With derived ids none is needed: +tallyman posts `/reload_expr/` for each entry of the project and treats +the 404 that Buckaroo returns for an unknown session as "not open", a case +the function already handles. That is one request per entry per klass change. +Found in the second review; the first draft of this decision did not mention +the reload. + Deleting a snapshot (D12) ends no session. A tab that already has the entry open fails on its next query, inside Buckaroo, with the `At least one path is required` error above. That is accepted: the user @@ -450,6 +506,12 @@ empty. The test fails on `main` today (finding 1), so it belongs in the failing-tests commit, and it outlives this change as the check that no xorq cache node has crept back in. +xorq reads `XORQ_CACHE_DIR` once, when it is first imported, and +`tests/conftest.py` points the whole test session at one directory for that +reason. So the sentinel cannot be a fresh directory per test. The test either +asserts that no file appears in the session's directory while it runs, or runs +its steps in a child process with a value of its own. + ### D9. One change, one rebuild Removing the cache node changes the hash of every worthy entry, and bare-read @@ -522,7 +584,9 @@ discover them: not change it. The lock is blocking and has no timeout, and the work it now covers is long: a -materialization runs single-partition and twice (ADR-009 decisions D1 and D6). +materialization runs single-partition, and a create runs its query twice +(ADR-009 decisions D1 and D6). A heal runs it once, since the recorded digest +is what it is compared against. A page request that needs a heal therefore waits behind any build in the other process. Paddy, 2026-09-20: correct first. #186 tracks the waiting. @@ -546,12 +610,123 @@ callers. There is no disk budget yet. That is the rewrite of between entries (`app.py:807-824`). That undoes the Cache page's delete button, and one large heal blocks startup for as long as it takes. - The verify sweep reads and never writes (D5). +- A reset does not touch `compute_cache/` (D14). - An explicit delete skips an entry marked not reproducible (decision D6 of `plans/ADR-009-digest-stability.md`), whose file cannot be recreated. The skip protects the file from the Cache page only. `compute_cache/` as a whole is still deletable by definition (D7), and where such a file should live is part of #185. +### D13. A file is cache only if `ensure_materialized` can re-create it + +Three classes of file sit behind an entry, and until now only the first had a +whole lifecycle: + +| File | Written by | If it is missing | So it is | +| --- | --- | --- | --- | +| Snapshot, `compute_cache/result_cache/.parquet` | `materialize` (D4) | re-run the entry's build and verify the digest (D5) | cache | +| Ordered copy of a source (decision D2 of `plans/ADR-008-row-order-of-reads.md`) | ingest, through polars | re-run ingest on the clone, with the reader options the manifest records | cache | +| Clone, `data/.cas/` | `ensure_cas_path` | copy the live source again, but only while its bytes still hash to the digest | data | + +The rule: a file is cache only if `ensure_materialized` can re-create it from +files that are not cache, and check what it made. Everything under +`compute_cache/` has to satisfy that, and then anything may delete it (D7, +D12). A file that is not cache is never deleted by machinery. + +What follows from the rule: + +- **Ordered copies are cache, so they live under `compute_cache/`**, inside + the project, where the portable-path placeholder covers them and the cold + test of D7 deletes them. `csv_ordered/` under `TALLYMAN_HOME` is retired. + For `ensure_materialized` to re-run ingest, the manifest's `sources` records + each source's reader options (the schema spec and the `scan_csv` options) + next to its digest. It holds only `{path: digest}` today. A re-created copy + is checked against a digest recorded when the copy was first written, which + is the rule D5 applies to snapshots. Where the copies live and the recorded + digest both follow from the rule and are not yet confirmed by Paddy in so + many words. They settle open question 3 here and open question 5 of + `plans/ADR-008-row-order-of-reads.md`. +- **Clones are data.** A clone is the only frozen copy of the bytes an entry + was built from, and it cannot be made again once the live file has been + edited. `gc_cas` deletes one today as soon as no surviving entry refers to + it, which D14 changes. `recon_cas_path` already re-clones a missing clone + from the live source when the live bytes still hash to the digest, but it is + reached only by re-running a recipe, which reads stopped doing in ADR-006. + That check moves into step 3 of D5. +- **The snapshot of an entry that is not reproducible** (decision D6 of + `plans/ADR-009-digest-stability.md`, create runs the query twice and + compares) is data that sits in the cache directory. Where it should live is + part of #185, as D12 already says. + +Measured on today's code (`scripts/spike_reset_roundtrip.py`). Step s1 holds +an entry over `orders.parquet`. Step s2 adds a cheap entry and a worthy entry +over `extra.parquet`, which is never touched: + +| After | Cheap entry | Worthy entry | +| --- | --- | --- | +| building s2 | reads | reads | +| a reset to s1, then a reset forward to s2 | `ValueError: At least one path is required` | reads, from the snapshot the bullpen gave back | +| the same, then `compute_cache/` emptied | the same error | the same error, from the heal | + +The reset to s1 deleted the clone of `extra.parquet`, the reset forward +restored the entries that read it, and nothing on the read path makes a clone +again. This is a defect in today's code and does not depend on this ADR. + +*Rejected:* keep ordered copies next to the clone under `data/.cas/`, as the +fix for #168 first proposed. `gc_cas` deletes every file in that directory +whose stem is not a live source digest, and an ordered copy is named by a key +and not by a digest, so every reset would delete every ordered copy. + +### D14. A reset leaves `compute_cache/` alone + +`reset_to` (`catalog_state.py:327`) returns the catalog to an earlier step +with `git reset --hard`, and then reconciles the files git does not track. For +`compute_cache/` it keeps a git-tracked list, `compute_cache.jsonl`, of every +file that was under the directory at each checkpoint +(`capture_tallyman_state`). A reset moves each file that is not on the target +step's list into the bullpen (`prune_compute_cache`), copies back each listed +file that is missing (`restore_from_bullpen`, a bare `shutil.copy2` onto the +final path), and then deletes every clone that no surviving entry refers to +(`gc_cas`). None of the three ADRs of this set mentioned it before the second +review. + +Against this ADR that machinery is a second owner of the cache: + +- the copy back is a second writer of snapshot files, outside D4. It is not + atomic, and since D5 treats "the file exists" as a hit, a page request can + open a half-copied file; +- the prune moves files that D12 says only the user deletes, including the + file of an entry that is not reproducible, which cannot be rebuilt; +- the list records whatever is under the directory, so it would pick up the + writer's temp files after a crash; +- capturing the list is a cost every checkpoint pays, and it grows with the + cache (#22). + +Decision: a reset stops managing `compute_cache/`. Snapshots are named by +content hash, so a file left behind by a retired entry cannot be served for +another entry. It is unreferenced disk until that entry comes back or the user +deletes it, and a snapshot that is missing after a reset is healed and +verified like any other (D5). `compute_cache.jsonl`, `prune_compute_cache` and +the `compute_cache/` half of `restore_from_bullpen` are deleted. + +The prune existed for one property: after a reset, an expression that is added +again should compute cold, so that a rehearsed demo shows real work. A create +always runs the query (D4), so that property holds without the prune. + +Clones are data (D13), so a reset stops deleting them. It moves a clone that +no surviving entry refers to into the bullpen, and a reset forward brings it +back, which is how a reset already treats entry directories. + +Three options were played through in the second review. Keeping today's +machinery needs an atomic, verified copy back and an exception for files that +cannot be rebuilt. Pruning without restoring keeps one writer and still needs +that exception. Leaving the directory alone needs neither. Paddy, 2026-09-20: +"I guess your suggestions sound good." + +Cost accepted: the snapshot of a retired entry stays on disk until the user +deletes it. The Cache page lists files by entry, so it needs a row for files +whose entry is not in the catalog. + ## Testing Every test below goes in the failing-tests commit and is seen red on CI before @@ -559,9 +734,8 @@ the change lands (D9, step 1). A test of a function that does not exist yet fails on import, and that counts as red. An earlier draft let such tests ride with the change. Paddy, 2026-09-20: do normal TDD. -- **Sentinel** (D8). With `XORQ_CACHE_DIR` pointing at an empty - directory, a build, a chained child build, a view, a delete and a reopen - leave that directory empty. +- **Sentinel** (D8). A build, a chained child build, a view, a delete and a + reopen write nothing under xorq's cache directory. - **Concurrent builds** (D11). Two threads building the same entry both return it, and the entry's directory is intact afterwards. - **Forgotten session** (D6). After Buckaroo has dropped a session, @@ -587,6 +761,19 @@ with the change. Paddy, 2026-09-20: do normal TDD. sweep reports the entry as absent, and the file is still absent afterwards. - **Forced reload** (D6). After an unfaithful heal, Buckaroo receives a `/load_expr` for that entry's session id with `force_reload` set. +- **Klass reload without a session record** (D6). After a klass is added, a + grid that is open shows it, and an entry that was never opened has no + session afterwards. +- **A create always runs** (D4, D14). Building an entry whose snapshot file is + already on disk runs the query and replaces the file. +- **Every class of file is re-created** (D13). With an ordered copy deleted, + opening a root entry makes it again from the clone and checks it. With a + clone deleted and the live source unchanged, the clone is made again. With + the live source changed as well, the error names the source file. +- **A reset back and forward breaks nothing** (D13, D14). After a reset to an + earlier step and a reset forward, every entry of the restored step reads, + cheap and worthy, and still reads after `compute_cache/` is emptied. The + reset moved and copied nothing under `compute_cache/`. ## Consequences @@ -597,7 +784,18 @@ with the change. Paddy, 2026-09-20: do normal TDD. and the `(FileNotFoundError, ValueError)` retry around xorq's shared temp file; `entry_graph_expr` as a separate function; tallyman's record of Buckaroo sessions (`_sessions`, `~/.tallyman/buckaroo_sessions.json`) and - `evict_session`, which worked by dropping an entry of it (D6). + `evict_session`, which worked by dropping an entry of it (D6); + `compute_cache.jsonl`, `prune_compute_cache` and the `compute_cache/` half + of `restore_from_bullpen` (D14); `csv_ordered/` under `TALLYMAN_HOME` (D13). +- **#77:** closed by D2, though not tested on a second machine. #77 is an + empty grid on a clone of the project at another path. The snapshot's name + came from xorq's tokenization of a graph that contains the build-time path, + so a heal on the new machine wrote its file under the key that machine + computed, while the child's frozen build read the key computed at build + time. Under D2 the name is the entry's recorded content hash, which is its + directory name and is the same on every machine, and the literal path in a + child's build goes through the `${TALLYMAN_PROJECT_ROOT}` placeholder. D5 + then heals the file where the build reads it. - **ADR-006:** its D4 (inlined chaining) and D8 (snapshot-key tripwire) are superseded. Its D2's "snapshot path derived from the loaded expression" becomes `snapshot_path`. Its D3 (rebind composition onto the default backend) @@ -627,38 +825,55 @@ with the change. Paddy, 2026-09-20: do normal TDD. - **Source identity `salt` mode:** `rewrite_for_build` returns early under `salt` because xorq's path-only snapshot keys would collide. A snapshot named by the entry's content hash has no such collision, so the early return should - become unnecessary. Not tested. + become unnecessary. Not tested. The early return also skips the canonical + sort, so while it stays a worthy entry built under `salt` is written in the + engine's order. - **Cost accepted:** a worthy parent's snapshot must exist before a child can be built. A build also no longer repairs its own ancestors when something outside tallyman executes it, and under the governing rule nothing does. A tab open on an entry whose file the user deletes errors until the entry is - reopened (D6). A missing ordered copy of a source surfaces as a Buckaroo - error (D5). + reopened (D6). The snapshot of an entry that a reset retired stays on disk + until the user deletes it (D14). ## Open questions -1. **Do two kinds of entry survive?** This is the largest open question of the - set and Paddy has not answered it. Materializing every entry when it is - created would remove the cheap and worthy classifier, the build error of - ADR-008 decision D3, the allow-list of ADR-008 decision D4, the view case in - D6, and open question 2 below, and it would give every entry a digest. An - entry built on another would always read the parent's file, so no graph - would be more than one entry deep, which is the parquet boundary Paddy wanted - in June. `plans/ADR-003-result-cache-cost-rubric.md` already proposes - admitting every result and evicting by budget. The cost is one file per - entry: one project measured 19 GB of cache for 779 MB of data while every - CSV revision wrote a file. His answer to question 4 ("materialize the - parquet if necessary") stands until he says otherwise. -2. **Deep cheap chains.** Nothing cuts the graph between cheap entries. Moot if - open question 1 is answered with one kind of entry. +1. **Do two kinds of entry survive?** Closed on 2026-09-20. Paddy: "yep keep + two kinds". A worthy entry is materialized when it is created, and a cheap + entry is a stored plan over files that exist (D6). The number is kept so + that references to open questions 2 and 3 stay valid. + + The alternative was to materialize every entry when it is created. It + would have removed the cheap and worthy classifier, the build error of + ADR-008 decision D3, the view case in D6 and open question 2, given every + entry a digest, and cut every graph at one entry deep, which is the parquet + boundary Paddy wanted in June. Its cost is one file per entry: one project + measured 19 GB of cache for 779 MB of data while every CSV revision wrote a + file. The second review measured that it was workable. On the + single-partition connection `materialize` uses, every row-preserving plan + streamed its rows in the parent file's order, three runs out of three, + against none on the default connection (`scripts/spike_stream_order.py`), + so the writer could have numbered rows in parent order with no column + carried through the recipe. + + What keeping two kinds costs is that the test for "cheap" becomes a + guarantee. A cheap entry has no file of its own and pages by its parent's + `__row_order`, so the test has to ensure that the column is still present + and still unique at the top of the plan. Four decisions of + `plans/ADR-008-row-order-of-reads.md` carry that, and each was tightened or + added in the second review: D3 (dropping the column is a build error), D4 + (the test itself), D6 (joins) and D12 (a raw parquet read is a build + error). +2. **Deep cheap chains.** Nothing cuts the graph between cheap entries. Still + open, now that open question 1 is closed with two kinds. 3. **Where ordered copies of sources live.** ADR-008 decision D2 adds one per parquet source. `csv_ordered/` is global today, is never collected, and is not included by `tallyman pack`. Its path is also outside the project root, so `make_portable_inplace` does not rewrite it and a CSV entry's build is - not portable. The fix for #168 proposes keeping a CSV's ordered copy under - the project, next to the content-addressed clone it is built from, which - would settle this for CSVs. If every entry is materialized, a root entry's - own file could serve as the ordered copy. + not portable. Settled in outline by D13: an ordered copy is cache, so it + lives under the project's `compute_cache/`, and `ensure_materialized` makes + it again from the clone when it is missing, so `tallyman pack` has no need + to ship it. Not yet confirmed by Paddy in so many words. What remains is + the directory's name. Closed in the grilling session: eviction policy (D12). diff --git a/plans/ADR-008-row-order-of-reads.md b/plans/ADR-008-row-order-of-reads.md index 6cdae2d..fae0ecb 100644 --- a/plans/ADR-008-row-order-of-reads.md +++ b/plans/ADR-008-row-order-of-reads.md @@ -2,8 +2,9 @@ - **Status:** Proposed (2026-09-18, revised 2026-09-20 in the grilling session, and again the same day after a review of PR #184, which made the - fix for #168 a precondition of D2 and D7). Awaiting Paddy's review; nothing - here is implemented. The first draft pinned row order with an engine setting. Paddy + fix for #168 a precondition of D2 and D7, and a third time that day after a + second review: D4 and D6 were tightened, and D10 to D12 are new). Awaiting + Paddy's review; nothing here is implemented. The first draft pinned row order with an engine setting. Paddy proposed baking a row-order column into every file tallyman writes and sorting every page by it. The measurements below favour that, so it is now the decision and the engine setting is the rejected alternative under D5. @@ -15,9 +16,14 @@ 0.15.4, xorq 0.3.26, xorq-datafusion 0.2.7). - **Tickets:** #168 (CSV sources bypass source identity; D2 and D7 depend on its fix), buckaroo-data/buckaroo#974 (Buckaroo's half of D5, see D8), #188 - (diffs, moved out of ADR-007). + (diffs, moved out of ADR-007), #12 (the classifier reads `expr.yaml` with a + regex, which D4 retires), #146 (ordering inside window functions and + ordered aggregates, which D10 leaves there). - **Affected code:** `src/tallyman_xorq/source_cache.py` (`rewrite_for_build`, - `_tie_break_order`), `src/tallyman_xorq/io.py` (`read_project_file`, + `_tie_break_order`, `_canonical_sorted`, `_is_worthy_expr`), + `src/tallyman_xorq/build.py` (`_csv_direct_read_check`, which gains a + parquet sibling, and the hint that turns ibis's name-collision error into + an instruction), `src/tallyman_xorq/io.py` (`read_project_file`, `tallyman_read_csv`, `io.py:627`, and for #168 `_ordered_csv_key` and `_ordered_csv_parquet`), `src/tallyman_xorq/source_identity.py` (the three steps a CSV now goes through), `src/tallyman_xorq/result_cache.py` @@ -36,8 +42,12 @@ - **Evidence:** `scripts/spike_row_order_paging.py` (the decisions), `scripts/spike_window_read_order.py` (the problem, and the rejected engine-setting approach), `scripts/spike_csv_source_identity.py` (D2 and D7: - what a CSV edit does to a content hash) and - `scripts/spike_deep_page_memory.py` (the memory figures under Consequences). + what a CSV edit does to a content hash), + `scripts/spike_deep_page_memory.py` (the memory figures under Consequences), + and four from the second review: `scripts/spike_cheap_classifier.py` (D4), + `scripts/spike_row_order_joins.py` (D6), `scripts/spike_sort_grafting.py` + (D10, D11 and the note on parquet statistics under D5) and + `scripts/spike_ordered_copy_layout.py` (D2 and open question 1). All figures are from those scripts on a 14-core machine. ## Terms @@ -55,6 +65,11 @@ computed columns, renames and casts qualify. Aggregates, joins, unions, distincts and unnests do not. - **Tie:** two or more rows with equal values in every sort key. +- **Natural order:** the order of rows in the file they came from, which + `__row_order` records (D2). +- **Graft:** to add sort keys to a query that its author did not write. +- **Hoist:** to take the keys of a sort from lower in a recipe and lead the + top-level sort with them (D11). - **Exchange operator:** a DataFusion physical-plan step (`RepartitionExec`, `CoalescePartitionsExec`) that moves rows between parallel partitions. After one, rows arrive in whatever order the partitions finish. @@ -150,7 +165,11 @@ Two writers produce it: steps `read_project_file` performs for parquet today, `io.py:78-84`), and the ordered copy is built from the content-addressed clone, which stays the immutable input. A source that already has a `__row_order` column has it - overwritten, which is the right outcome for a file tallyman exported. + overwritten, which is the right outcome for a file tallyman exported. The + copy lives under the project's `compute_cache/`, and `ensure_materialized` + makes it again from the clone when it is missing (decision D13 of + `plans/ADR-007-tallyman-owned-materialization.md`, which files are cache). + D12 closes the one way a parquet file could enter without a copy. CSVs have the ordered-copy step today and not the keying. The first draft of this decision said they already worked this way, which was wrong. The @@ -167,7 +186,8 @@ The canonical sort's tie-break (`_tie_break_order`) puts an inherited `__row_order` where `original_row_order` is today: after the author's own `order_by` keys and before the remaining columns. A worthy entry that keeps its parent's rows, such as one adding a window function, therefore keeps the -parent's order. +parent's order. D10 applies the same tie-break to every sort in a recipe, not +only to the last one. *Rejected:* `row_number()` inside the entry's graph. It needs the same global sort, adds a window function to every worthy build, and leaves contiguity to @@ -191,8 +211,10 @@ descriptions say the same thing up front. This is the feedback channel A worthy entry is exempt, because the writer numbers its rows (D2). An author changes `__row_order` by asking for an order: `order_by` makes the entry worthy, -and the writer numbers the rows in the requested order. Assigning to the column -directly stays an error (D6). +and the writer numbers the rows in the requested order. That holds for an +`order_by` anywhere in the recipe only because of D11. As first drafted it was +true of an `order_by` that is the recipe's last step, and of no other. +Assigning to the column directly stays an error (D6). Tallyman makes one alteration of its own, at the top of the expression only: it moves `__row_order` to the last position, since a computed column added after @@ -220,15 +242,63 @@ A cheap entry inherits its row order, so it must be a row-preserving plan over exactly one file. The classifier changes from a deny-list to an allow-list: an entry is cheap only if every relation operation in its graph is known to be row-preserving (a file read, a filter, a column selection, a computed column, a -rename, a cast, a column drop). Anything else is worthy, including operations -nobody has thought about yet. Today's `_EXPENSIVE_OPS` deny-list classes -`Union`, `Distinct` and `Unnest` as cheap, and none of them can carry one -parent's row order. +rename, a cast, a column drop, a drop of null rows, a fill of nulls). Anything +else is worthy, including operations nobody has thought about yet. Today's +`_EXPENSIVE_OPS` deny-list classes `Union`, `Distinct` and `Unnest` as cheap, +and none of them can carry one parent's row order. + +A list of relation operations is not enough, which the second review measured +(`scripts/spike_cheap_classifier.py`). Paddy confirmed on 2026-09-20 that two +kinds of entry stay (open question 1 of +`plans/ADR-007-tallyman-owned-materialization.md`), so this test is what +guarantees that a cheap entry pages repeatably, and it has to look inside the +allowed operations as well. The parent has 6 rows: + +| Recipe shape | Relation operations | Today's deny-list | Relation allow-list alone | Rows | `__row_order` unique | +| --- | --- | --- | --- | --- | --- | +| `t.select("k", "__row_order", tag=t.tags.unnest())` | read, select | cheap | cheap | 8 | no | +| `t.mutate(rn=ibis.row_number())` | read, select | worthy | cheap | 6 | yes | +| `t.mutate(prev=t.v.lag())` | read, select | worthy | cheap | 6 | yes | +| `t.mutate(r=ibis.random())` | read, select | cheap | cheap | 6 | yes | +| `t.filter(t.k.isin(u.k))` | two reads, filter, select | cheap | cheap | 3 | yes | + +An `unnest` written inside a select is an ordinary column selection to a list +of relation operations. It multiplied the rows and duplicated `__row_order`, +which breaks the "no ties" that D5 depends on. A window function inside a +computed column is an ordinary selection too, and its values depend on the +order the rows arrive in (D10). + +So the test has three parts, each decided on the live expression by the class +of the operation: + +- every relation operation is on the list above; +- the plan reads exactly one file; +- no value operation multiplies rows (`Unnest`), depends on row order + (`WindowFunction`, which also covers `row_number`, `lag` and a total used + inside a computed column), or is not pure (`Impure`, which is `random()` and + `uuid()`; `now()` and `today()` by name, since xorq's ibis classes them as + constants; and any UDF, matched as it is today). + +The verdict is computed once, when the entry is built, and recorded in the +manifest. `result_cache.cache_worthy()` and the gate on primary-key +inheritance (`primary_key.py:204`) read the manifest and stop classifying. +`classify_build` and its regex over `expr.yaml` are retired, which closes #12. +A regex cannot hold an allow-list: `op:` in that file also matches `DataType`, +`FrozenDict`, `float` and `tuple`, which are not operations, and `Field` and +`Literal`, which are not relations. There is then one implementation, and +nothing to keep in lockstep. A new or unknown operation now costs a copy (safe) instead of unstable paging -(unsafe). `classify_build` (which reads the serialized build) and -`_is_worthy_expr` (which reads the live expression) flip together, as they must -today. +(unsafe). So does a second file, and so does `random()`. + +The three parts are what keeping two kinds requires, as proposed in the second +review, and Paddy has not confirmed the list in so many words. The "not pure" +part reaches into #185 (non-pure recipes). It makes a recipe that calls +`random()` worthy, so decision D6 of `plans/ADR-009-digest-stability.md` +(create runs the query twice) checks it and pins its file, and the cheap half +of #185 has nothing left to decide. If that is unwanted, striking `Impure` and +the two names restores today's behaviour, where such an entry re-runs on every +read. Supporting measurement from the first draft: a union of two files returned 2 different pages for 8 identical requests even on a single-partition @@ -241,7 +311,8 @@ operator merges them. - User sort: the user's keys, then `__row_order` ascending as the last key, which breaks every tie. -That is the whole rule, for every entry and both processes. Two faster paths +That is the whole rule, for every entry and both processes. It is the +page-request half of Paddy's rule in D10. Two faster paths exist and are deliberately not part of this decision (Paddy, 2026-09-20: a cohesive system that works reliably comes first, and speed problems are handled as they come up): @@ -254,6 +325,18 @@ as they come up): Both are measured below so the numbers are on hand when they are wanted. +A third was asked about in the second review: leaving the last key off when +the user's sort key is already unique. Parquet's statistics cannot show that. +pyarrow writes a minimum, a maximum and a null count for each column chunk and +no distinct count, so a unique column and one holding a value twice have the +same statistics (`scripts/spike_sort_grafting.py`), and a distinct count would +be per row group in any case. Tallyman's writer could record the fact itself, +since rows reach it sorted and ties on the author's keys are adjacent. It +would save little: when the leading key is unique the comparison never reaches +`__row_order`, and when it is not, the key is needed. It could apply to page +requests only, because the keys D10 adds at build time are part of the hashed +graph, and it would have to be repeated inside Buckaroo. Not planned. + Measured on 3,000,000 rows by 14 columns (287 MB), on the default parallel connection with no engine settings. Every row of the table returned the correct page 6 times out of 6: @@ -301,12 +384,25 @@ semantics, in any engine and any process. - A recipe may read the column and may copy it under another name (D3). A recipe that assigns to `__row_order` is a build error, because arbitrary values could contain ties or gaps, and D5 depends on `0..N-1` with neither. -- Only the exact name is special. `__row_order_v1`, or any other name an author - picks for a copy, is ordinary data and survives materialization. +- Only the exact name is special, and ibis's collision name for it, which the + next item covers. `__row_order_v1`, or any other name an author picks for a + copy, is ordinary data and survives materialization. - A join of two entries leaves the right side's copy behind under ibis's - collision name, `__row_order_right`, and a three-way join silently keeps only - the first two. That column is ordinary data too: it says where the row sat in - the right-hand parent. The writer replaces only `__row_order` itself. + collision name, `__row_order_right`. The writer drops that one column, as + well as replacing `__row_order`. An earlier draft kept it as ordinary data + and said that a three-way join "silently keeps only the first two". Both + were wrong (`scripts/spike_row_order_joins.py`). A three-way join written in + one recipe shows the expected columns and passes `build_expr`, then raises + `IntegrityError: Name collisions: {'__row_order_right'}` from the canonical + sort, which is the next build step. A join entry that kept the column fails + the same way as soon as it is joined to a third entry, and with the column + dropped from its file that second join builds. For a three-way join in one + recipe the author has to drop `__row_order` from the right-hand inputs, and + the build turns ibis's message into that instruction, since the author never + wrote the name it complains about. An author who wants the right-hand + positions keeps them under a name of their own, as D3 describes. Proposed in + the second review as part of what keeping two kinds requires, and not yet + confirmed by Paddy in so many words. - The primary-key search skips it. Nothing excludes `original_row_order` from the candidates today (`primary_key.py:219`), and a column that is unique in every table would win the search for any table without a string or id key. @@ -344,7 +440,7 @@ What INV-2 provided, and what replaces it: | A canonical display order | D5. INV-2 did not deliver this above 10 MB. | | A parquet boundary for chained children | A cheap root's graph is one read node. | | Fixed rows under the root's hash, through its baked snapshot | The content-keyed ordered copy of D2, written once. Requires the fix for #168. | -| A `result_digest` on the root, so a re-parse that produced different rows would be caught | Lost as it stands: cheap entries record no digest (ADR-006 decision D9, "no cheap-entry digests"). It matters only when an ordered copy is deleted and re-created. See open question 5. | +| A `result_digest` on the root, so a re-parse that produced different rows would be caught | A digest recorded for the ordered copy itself, which a re-created copy is checked against (ADR-007 decision D13, which files are cache). Cheap entries still record no digest of their own (ADR-006 decision D9, "no cheap-entry digests"). Not yet confirmed; see open question 5. | Every hash in every CSV lineage changes, so this rides the corpus rebuild of ADR-007 decision D9 ("one change, one rebuild"). @@ -376,6 +472,135 @@ does not establish a split scan and is not what makes that test meaningful; its aggregate is), and `src/tallyman_xorq/source_cache.py:98`. `tests/test_tallyman_read_csv.py:159` already says about 10 MB. +### D10. The natural order is imposed on every `order_by` + +Paddy's rule, 2026-09-20: "every query should have a unique sortby clause, if +the base query didn't have one, it needs to be grafted onto the query so that +everything else is order deterministic", and then: "when the user/mcp supplied +sort isn't deterministic, impose the natural order into each order by so the +resulting query becomes deterministic." + +The reading that was put to him the same day, which he did not overrule: + +- **Always.** Tallyman cannot know whether a supplied sort is unique without + scanning the table (D5 records why parquet statistics do not help), and a + last key added to a sort that is already unique changes nothing. +- **The natural order, then the remaining sortable columns.** The natural + order alone is unique only while rows come one for one from a single file + tallyman wrote. Above a join or a union it has ties, above an outer join it + has nulls, a value-level `unnest` duplicates it (D4), and after an aggregate + it is gone. With the remaining columns after it the sort is total up to rows + that are identical in every sortable column, and those write the same bytes + in either order. This is the tie-break `_tie_break_order` already builds. +- **At every sort in the recipe, and on every page request.** Page requests + are D5. The build-time half is new: `_canonical_sorted` extends an author's + sort only when it is the top node of the expression + (`source_cache.py:113`), and leaves every other `Sort` node as written. + +Why this has to reach every sort: a sort that feeds a `limit` decides which +rows the entry holds, and the top of the expression is too late to break its +ties. `order_by(g).limit(1000)` over 3,000,000 rows, where about 15,000 +rows tie on the smallest `g`, five runs each +(`scripts/spike_sort_grafting.py`): + +| Connection | Sort key | Distinct sets of rows in 5 runs | Equals the rows `(g, id)` picks | +| --- | --- | --- | --- | +| default | `g` | 5 | no | +| `target_partitions = 1` | `g` | 1 | yes | +| default | `g`, then the row position | 1 | yes | + +The middle row is how this case works today. It is repeatable because +materialization runs single-partition (decision D1 of +`plans/ADR-009-digest-stability.md`), which is the engine's behaviour and the +kind of dependence D5 rejects for page requests. The last row is repeatable by +the query's own meaning, on any connection. A cheap entry cannot contain a +sort, because a sort makes an entry worthy (D4), so the build-time half only +ever runs at materialization. + +The added keys are part of the build's graph, so the content hash covers them, +and this rides the corpus rebuild of ADR-007 decision D9 ("one change, one +rebuild"). + +Left to #146 (a lint for row-ordering nondeterminism): the `order_by` inside a +window function, and ordered aggregates such as `first`, `last` and `collect`. +They are the same rule. `cumsum()` with no order gave 5 distinct sets of +values in 5 runs on the default connection and 1 on a single-partition one. +Ordered by the tied key `g`, and by `g` then the row position, it gave 1 on +both. An entry with a window function is always materialized, so it is +repeatable today, by the engine's behaviour again. Only `cumsum()` was +measured. + +*Rejected:* add the keys only when the supplied sort is not unique. That needs +a scan of the table for every sort, to save a key that costs almost nothing +when the leading key is unique. + +### D11. A sort that is not the recipe's last step is hoisted, or the build fails + +`_canonical_sorted` recognizes an author's sort only when it is the top node. +When another step follows, the check fails and the whole expression is wrapped +in a sort that leads with the inherited row order. That column is unique, so +the author's sort has no effect on what is written. The parent's rows are in +the order 40, 10, 60, 20, 50, 30 and the author asks for `amount` descending +(`scripts/spike_sort_grafting.py`): + +| Recipe | Classed | Written as | +| --- | --- | --- | +| `order_by` last | worthy | 60, 50, 40, 30, 20, 10 | +| `order_by`, then `mutate` | worthy | 40, 10, 60, 20, 50, 30 | +| `order_by`, then `select` | worthy | 40, 10, 60, 20, 50, 30 | +| `order_by`, then `filter(amount > 15)` | worthy | 40, 60, 20, 50, 30 | +| `order_by`, then `limit(3)` | worthy | 40, 60, 50 | + +Each of these is classed worthy because of that sort, so the author pays for a +full copy that ignores it, and a top-three entry is not shown in rank order. +Under Paddy's rule (D10) the author did supply a sort, so the system keeps it: + +- From the top of the expression, walk down through steps that keep the order + of rows (a selection, a computed column, a filter, a limit, a column drop, a + rename, a cast, a drop of null rows, a fill of nulls) to the nearest sort. +- When each of that sort's keys is still an output column, unchanged (a rename + is followed), the top-level sort leads with those keys, then the tie-break + of D10. +- When a key did not survive, because it was dropped, overwritten, or was an + expression and not a column, the build fails. The error names the key and + tells the author to keep the column or to sort as the last step. This is the + choice D3 makes for a dropped `__row_order`: report what the author can fix + in one line, and do not repair it silently. + +Paddy, 2026-09-20: "I like your suggestion." The change is about 40 lines in +`_canonical_sorted`. + +*Rejected:* make any `order_by` that is not the last step a build error. It is +one rule, but a top-N recipe would then have to be written +`order_by(...).limit(n).order_by(...)`. + +### D12. A raw parquet read is a build error + +D2 says every file tallyman reads carries `__row_order`, and that both kinds +of source go through source identity first. Neither is enforced for parquet. +A recipe can call `xo.deferred_read_parquet(abs_path)`, and tallyman's own +hints recommend it (`build.py:199`, `source_cache.py:133`, and the namespace +note in a tool description, `src/tallyman_mcp/server.py:275`). Such a read has no digest, no clone and +no `manifest.sources` record, which is the defect of #168 for a parquet file. +It also has no ordered copy, so a root entry built on it has no `__row_order` +to page by. Only the CSV form is banned today (`_csv_direct_read_check`). + +A read of a parquet file that tallyman did not write becomes a build error +that points the author to `read_project_file`, next to the CSV check, and the +three hints change with it. Tallyman's files are the snapshots and ordered +copies under the project's `compute_cache/` (decision D13 of +`plans/ADR-007-tallyman-owned-materialization.md`, which files are cache), so +the check is on the path of each `Read`. Nine files under `tests/` call +`deferred_read_parquet` today, 23 calls in all, so the change carries test +churn. + +Proposed in the second review as part of what keeping two kinds requires, and +not yet confirmed by Paddy in so many words. + +*Rejected:* rewrite a raw read into an ingest at build time. It is the surgery +inside an author's expression that D3 turned down, and the recipe text would +name one file while the entry read another. + ## Testing Every test below goes in the failing-tests commit and is seen red on CI before @@ -410,15 +635,38 @@ that does not exist yet fails on import, and that counts as red. Paddy, - **CSV roots** (D7). A `tallyman_read_csv` entry has no Sort in its build, is classed cheap, and has exactly one row-order column. - **The hint** (D8). The `/load_expr` payload names `__row_order`. +- **The cheap test looks inside** (D4). A value-level `unnest`, a window + function in a computed column, `random()` and a filter against a second file + are each classed worthy, a drop of null rows is classed cheap, and the + verdict is read from the manifest with no `expr.yaml` parsed. +- **Joins** (D6). A join entry's file has no `__row_order_right`, joining it to + a third entry builds, and a three-way join in one recipe fails with a + message that says to drop `__row_order` from the right-hand inputs. +- **Every sort is total** (D10). An `order_by(g).limit(k)` recipe over a file + above the split threshold, with ties on `g` at the cut, holds exactly the + rows that `(g, __row_order)` picks. +- **A sort that is not last is kept** (D11). A top-three recipe is written in + rank order, `order_by` then `mutate` is written in the sorted order, and a + recipe that drops its sort key in a later select fails to build with a + message that names the key. +- **Raw reads** (D12). A recipe that calls `xo.deferred_read_parquet` on a + source file fails to build with a message that names `read_project_file`. ## Consequences - Pages are repeatable for unsorted and sorted requests, in tallyman and in Buckaroo, with no engine settings and no second connection. -- Every table shows one more column, at the end. A join result also shows - `__row_order_right` unless the recipe drops it. +- Every table shows one more column, at the end. A join result does not carry + `__row_order_right`, because the writer drops it (D6). - A recipe whose select list forgets `__row_order` fails to build until the - author adds it. + author adds it. So does a three-way join that keeps the column on its + right-hand inputs (D6), a recipe that drops the key of a sort it made + earlier (D11), and a recipe that reads a parquet file directly (D12). +- Every sort in a recipe gains keys its author did not write (D10), and a sort + that is not the last step now decides the order that is written (D11). +- Whether an entry is cheap is decided once, at build, and read from the + manifest afterwards (D4). More recipes are worthy than before: one that + unnests inside a select, reads a second file, or calls `random()`. - CSV lineages stop writing sorted copies. With ADR-007, revisions of a CSV entry are cheap reads over one intermediate file, and primary-key inheritance applies to them. @@ -444,9 +692,12 @@ that does not exist yet fails on import, and that counts as red. Paddy, 1. **Ordered copies of parquet sources.** D2 adds a copy per parquet source. Adopted as the uniform rule under Paddy's "cohesive first" priority, and not - yet confirmed by him in so many words. It also assumes polars numbers a - parquet scan's rows in file order, as ADR-004 measured for CSV, which needs - checking. + yet confirmed by him in so many words. With two kinds of entry kept, a + cheap root pages by this column, so the copy is needed. It assumed polars + numbers a parquet scan's rows in file order, as ADR-004 measured for CSV. + Checked in the second review (`scripts/spike_ordered_copy_layout.py`, polars + 1.40.1): `scan_parquet().with_row_index()` numbered a source of 184 row + groups in file order on 1, 3 and 14 threads. 2. **Renaming `original_row_order`.** D7 replaces it with `__row_order`. Adopted on the same basis, and also not yet confirmed. The alternative keeps it as a data column meaning "line of the source file", at the cost of two identical columns on every CSV root. @@ -462,6 +713,9 @@ that does not exist yet fails on import, and that counts as red. Paddy, Recording the copy's digest in the root entry's manifest, and verifying it on re-creation, would cover that. It wants ADR-009's digest definition, and it touches where the copies live: `csv_ordered` is global, is never - collected, and is not packed (ADR-007 open question 3). Nothing re-creates a - missing copy automatically yet, which ADR-007 decision D5 (one entry point - makes files exist) records as an accepted gap. + collected, and is not packed (ADR-007 open question 3). The second review + made re-creation a designed path and not a gap: ADR-007 decision D13 (which + files are cache) puts the copies under the project's `compute_cache/`, has + `ensure_materialized` make a missing one again from the clone, and records + the digest the new copy is checked against. That answer follows from + D13's rule and is not yet confirmed by Paddy in so many words. diff --git a/plans/ADR-009-digest-stability.md b/plans/ADR-009-digest-stability.md index 1003329..fa92869 100644 --- a/plans/ADR-009-digest-stability.md +++ b/plans/ADR-009-digest-stability.md @@ -5,7 +5,10 @@ `plans/ADR-008-row-order-of-reads.md`, D1 lost its speed gate, and D6 is new; and again the same day after a review of PR #184: D1 and D3 now say what single-partition execution leaves undetermined, and D6's cheap-entry - half moved to #185). Awaiting Paddy's review; nothing here is implemented. Amends + half moved to #185; and a third time that day after a second review: D1 now + says how a loaded build gets onto the single-partition connection, and D3's + format version covers the ordered copies of sources). Awaiting Paddy's + review; nothing here is implemented. Amends `plans/ADR-004-result-digest-canonical-ordering.md` (Option A's "hash the snapshot bytes") and decision D5 of `plans/ADR-006-read-path-loads-builds.md` (the canonical sort), which said @@ -34,7 +37,10 @@ - **Evidence:** `scripts/spike_float_aggregate_digest.py`, `scripts/spike_logical_digest.py`, and `scripts/spike_float_layout_digest.py` (D1 and D3: what the layout of the - parent file does to a float total). + parent file does to a float total), and two from the second review: + `scripts/spike_single_partition_loaded_build.py` (D1: a loaded build has to + be rebound) and `scripts/spike_ordered_copy_layout.py` (D3: the layout + polars writes). ## Terms @@ -120,6 +126,31 @@ layout (see "What this does not fix" below). It is a separate connection from the default backend that serves page reads, because a long materialization must not share a context with them. +Making that connection is not enough. `materialize` executes a build that +`load_expr` loaded, and `load_expr` makes its own backend objects, so a loaded +build ignores a single-partition connection it was never bound to +(`scripts/spike_single_partition_loaded_build.py`, a float `SUM` and `AVG` +group-by over 3,000,000 rows): + +| How the loaded build is executed | Distinct digests in 5 runs | +| --- | --- | +| as loaded | 5 | +| as loaded, while a single-partition connection exists on the side | 5 (the build's own backend reports 14 partitions) | +| rebound onto that connection with `replace_sources` | 1 | +| `SET` applied to each backend the load made | 1 | + +Either of the last two works, and neither changes the process default +backend, which still reports 14. `materialize` rebinds. That is what +`_rebind_to_default_backend` (ADR-006 decision D3, rebind composition onto the +default backend) already does, aimed at the materialization connection. + +The same connection is why a bare `limit`, or a window function with no order, +is repeatable at materialization: on it a row-preserving plan streams its rows +in the parent file's order (`scripts/spike_stream_order.py`). That is the +engine's behaviour and not the query's meaning, which is what decision D10 of +`plans/ADR-008-row-order-of-reads.md` (the natural order is imposed on every +`order_by`) is for. + Cost: about 3x on the spike's aggregate. ADR-004 measured a 3.1M-group aggregate at 0.5 s parallel against 3.4 to 3.9 s single-partition, and a full 43-column read at 2.5 s against 6.9 s, on the 11.8M-row parking file. A @@ -262,6 +293,19 @@ connection's `batch_size` are therefore part of the reproducibility contract: the manifest records a snapshot format version that stands for both, and changing either is a corpus rebuild. +The same holds for the ordered copy of a source (ADR-008 decision D2), which +polars writes and this writer does not. Its layout is pinned separately, in +`_CSV_PARQUET_WRITE` (`io.py:91-96`, row groups of 122,880 rows), and an entry +that totals a float column straight from a source reads that layout. The +format version covers those settings too. The second review checked that +polars honours them: `sink_parquet` wrote full row groups of exactly 122,880 +rows, with an identical layout, on 1, 3 and 14 threads, from a CSV source and +from a parquet one (`scripts/spike_ordered_copy_layout.py`, polars 1.40.1). + +The size is fixed in rows and not in bytes. A table with long text columns +therefore holds a large row group in memory while it is written, and the +contract above means the size cannot be tuned for one table. + ### D4. A mismatch record names its likely cause With D1 and D2 in place the causes left are the recipe's own nondeterminism @@ -311,7 +355,11 @@ materialized entry is created: `plans/ADR-007-tallyman-owned-materialization.md` all rely on. Paddy moved it to #185 on 2026-09-20. Until that is settled a cheap entry that calls `random()` behaves as it does today: the #88 lint warns, and the entry - re-runs on every read. + re-runs on every read. The second review proposed a smaller answer, recorded + in ADR-008 decision D4 (the test for cheap) and not yet confirmed by Paddy: + an operation that is not pure makes an entry worthy. Such a recipe is then + checked here like any other materialized entry, and no cheap entry calls + `random()`. The cost is a second execution of every create of a materialized entry. It is accepted under the priority recorded in ADR-007 ("a cohesive system that works @@ -325,9 +373,10 @@ file survives. The pin protects the file from the Cache page's delete and from nothing else, since `compute_cache/` is deletable by definition (ADR-007 decision D7, the cold state is an empty `compute_cache`). -A heal is still verified against the recorded digest, as now. After D6 a -mismatch there means something changed underneath a reproducible entry, which -D4 attributes. +A heal runs the query once and is verified against the recorded digest, as +now. Only a create runs it twice, since a create has nothing recorded to +compare against. After D6 a mismatch at a heal means something changed +underneath a reproducible entry, which D4 attributes. ## Testing @@ -351,8 +400,14 @@ that does not exist yet fails on import, and that counts as red. Paddy, - **Create passes a reproducible recipe.** A deterministic recipe is recorded as reproducible, and the query is observed to run exactly twice. - **The layout is pinned** (D1, D3). Every snapshot has row groups of 1,048,576 - rows, the materialization connection reports the pinned `batch_size`, and - the manifest records the snapshot format version. + rows, every ordered copy of a source has row groups of 122,880 rows, the + materialization connection reports the pinned `batch_size`, and the manifest + records the snapshot format version. +- **The build runs on the single-partition connection** (D1). While + `materialize` executes a loaded build, every backend the plan touches reports + `target_partitions = 1`, and the process default backend does not. +- **A heal runs once** (D6). Reopening an entry whose file was deleted runs its + query exactly once. - **A pinned file survives an explicit delete.** The Cache page's delete skips it and says why. - **The schema comes from the file.** An entry with a `timestamp[s]` column @@ -381,7 +436,8 @@ that does not exist yet fails on import, and that counts as red. Paddy, - The same function can digest an ordered copy of a source (ADR-008, open question 5). - The row-group size and the materialization connection's `batch_size` are - frozen for the life of the corpus, and changing either is a rebuild (D3). + frozen for the life of the corpus, and so are the settings polars writes an + ordered copy of a source with. Changing any of them is a rebuild (D3). ## Open questions diff --git a/scripts/spike_cheap_classifier.py b/scripts/spike_cheap_classifier.py new file mode 100644 index 0000000..3cc5f1d --- /dev/null +++ b/scripts/spike_cheap_classifier.py @@ -0,0 +1,108 @@ +"""ADR-008 evidence (D4): what a test for "cheap" has to look at, now that two kinds of entry stay. + +A cheap entry has no file of its own and pages by its parent's ``__row_order``, so the test has to guarantee that the +column stays unique through the plan. D4's first wording was an allow-list of RELATION operations. This script runs +three tests over the same recipe shapes: + +- today's deny-list (``result_cache.classify_build``, a regex over ``expr.yaml``); +- a relation-only allow-list, which is D4 as first worded, with drop-null and fill-null added to its list; +- the test D4 now specifies: the relation allow-list, exactly one file read, and no value operation that multiplies + rows, depends on row order, or is not pure, decided by base class on the live expression. + +For each shape it also executes the plan and reports whether ``__row_order`` is still unique. + + uv run python scripts/spike_cheap_classifier.py +""" + +from __future__ import annotations + +import os +import re +import tempfile +from pathlib import Path + +HOME = Path(tempfile.mkdtemp(prefix="spike_cheap_classifier_")) +os.environ["XORQ_CACHE_DIR"] = str(HOME / "_global_xorq") # must be set before xorq is imported + +import pyarrow as pa # noqa: E402 +import pyarrow.parquet as pq # noqa: E402 +import xorq.api as xo # noqa: E402 +import xorq.vendor.ibis as ibis # noqa: E402 +import xorq.vendor.ibis.expr.operations as ops # noqa: E402 +from xorq.common.utils.graph_utils import walk_nodes # noqa: E402 +from xorq.expr.relations import Read # noqa: E402 +from xorq.ibis_yaml.compiler import build_expr # noqa: E402 +from xorq.vendor.ibis.expr.operations.core import Node # noqa: E402 + +from tallyman_xorq.result_cache import classify_build # noqa: E402 + +ROW_PRESERVING_RELATIONS = (Read, ops.Filter, ops.Project, ops.DropColumns, ops.DropNull, ops.FillNull) +NEVER_CHEAP_VALUES = (ops.Unnest, ops.WindowFunction, ops.Impure) +NEVER_CHEAP_BY_NAME = {"TimestampNow", "DateNow"} # these are Constant, not Impure, in xorq's ibis + + +def relation_only_allow_list(expr) -> bool: + relations = [n for n in walk_nodes((Node,), expr) if isinstance(n, ops.Relation)] + return all(isinstance(n, ROW_PRESERVING_RELATIONS) for n in relations) + + +def is_cheap(expr) -> bool: + nodes = list(walk_nodes((Node,), expr)) + relations = [n for n in nodes if isinstance(n, ops.Relation)] + if len({n for n in relations if isinstance(n, Read)}) != 1: + return False + if not all(isinstance(n, ROW_PRESERVING_RELATIONS) for n in relations): + return False + for n in nodes: + if isinstance(n, NEVER_CHEAP_VALUES) or type(n).__name__ in NEVER_CHEAP_BY_NAME: + return False + if any("UDF" in base.__name__ for base in type(n).__mro__): + return False + return True + + +def main() -> None: + n = 6 + tags = [["x", "y"], ["z"], [], ["x"], ["y", "z", "w"], ["q"]] + parent = {"k": list(range(n)), "v": [1.5, 2.5, None, 4.5, 5.5, 6.5], "tags": tags, "__row_order": list(range(n))} + pq.write_table(pa.table(parent), HOME / "t.parquet") + pq.write_table(pa.table({"k": [1, 3, 5], "__row_order": [0, 1, 2]}), HOME / "u.parquet") + t = xo.deferred_read_parquet(str(HOME / "t.parquet")) + u = xo.deferred_read_parquet(str(HOME / "u.parquet")) + + shapes = { + "filter + computed column": t.filter(t.k > 0).mutate(w=t.v * 2), + "rename, cast, drop a column": t.rename(key="k").mutate(v=t.v.cast("float32")).drop("tags"), + "drop_null, fill_null": t.drop_null(["v"]).fill_null({"v": 0.0}), + "unnest inside a select": t.select("k", "__row_order", tag=t.tags.unnest()), + "row_number() in a mutate": t.mutate(rn=ibis.row_number()), + "share of total in a mutate": t.mutate(share=t.v / t.v.sum()), + "lag() in a mutate": t.mutate(prev=t.v.lag()), + "random() in a mutate": t.mutate(r=ibis.random()), + "filter by membership in a second file": t.filter(t.k.isin(u.k)), + "filter against a scalar subquery": t.filter(t.v > t.v.mean()), + "limit": t.limit(3), + "distinct": t.select("k", "__row_order").distinct(), + } + print(f"parent has {n} rows\n") + print(f"{'recipe shape':40s} {'today':8s} {'relations only':15s} {'D4':7s} rows __row_order unique") + names: set[str] = set() + for label, expr in shapes.items(): + build = Path(build_expr(expr, builds_dir=HOME / "builds")) + names |= {m for p in build.glob("*.yaml") for m in re.findall(r"op:\s*([A-Za-z_]+)", p.read_text())} + today = "worthy" if classify_build(build)["worthy"] else "cheap" + relations = "cheap" if relation_only_allow_list(expr) else "worthy" + proposed = "cheap" if is_cheap(expr) else "worthy" + out = expr.execute() + print(f"{label:40s} {today:8s} {relations:15s} {proposed:7s} {len(out):4d} {out['__row_order'].is_unique}") + + relation_names = {c.__name__ for c in ops.Relation.__subclasses__()} | {"Read"} + not_relations = sorted(n for n in names if n not in relation_names and not hasattr(ops, n)) + print(f"\nnames the classify_build regex matches that are not operations at all: {not_relations}") + print( + f"value operations it matches alongside the relations: {sorted(n for n in names if n in ('Field', 'Literal'))}" + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/spike_ordered_copy_layout.py b/scripts/spike_ordered_copy_layout.py new file mode 100644 index 0000000..5d70873 --- /dev/null +++ b/scripts/spike_ordered_copy_layout.py @@ -0,0 +1,100 @@ +"""ADR-008 (D2, open question 1) and ADR-009 (D3) evidence: the ordered copy of a source, which polars writes. + +The ordered copy is not written by the snapshot writer of ADR-009 D3. An ungrouped float total depends on the layout +of the file it reads (#187), so the layout of an ordered copy is part of the reproducibility contract too. + +Questions, asked with polars running on 1, 3 and the default number of threads: + +1. CSV source: are the row groups exactly the size asked for, and is the layout the same on every thread count? +2. CSV source: is the row index 0..N-1 in file order? +3. Parquet source with many row groups: does ``scan_parquet().with_row_index()`` number the rows in FILE order? + ADR-008 open question 1 said this needed checking. + +polars reads ``POLARS_MAX_THREADS`` when it is imported, so each thread count runs in a child process. + + uv run python scripts/spike_ordered_copy_layout.py +""" + +from __future__ import annotations + +import hashlib +import json +import os +import subprocess +import sys +import tempfile +from pathlib import Path + +N = 1_500_000 +WRITE = {"compression": "zstd", "compression_level": 3, "row_group_size": 122880, "statistics": True} # io.py + + +def child() -> None: + import numpy as np + import polars as pl + import pyarrow as pa + import pyarrow.parquet as pq + + home = Path(tempfile.mkdtemp(prefix="spike_ordered_copy_")) + rng = np.random.default_rng(7) + positions = np.arange(N) + + def layout(path: Path) -> tuple[list[int], str]: + md = pq.ParquetFile(path).metadata + sizes = [md.row_group(i).num_rows for i in range(md.num_row_groups)] + return sizes, hashlib.sha256(repr(sizes).encode()).hexdigest()[:12] + + def ordered_copy(scan, out: Path) -> None: + indexed = scan.with_row_index("__row_order") + indexed.select(["id", "v", pl.col("__row_order").cast(pl.Int64)]).sink_parquet(str(out), **WRITE) + + csv = home / "src.csv" + pl.DataFrame({"id": positions, "v": rng.normal(size=N)}).write_csv(csv) + from_csv = home / "from_csv.parquet" + ordered_copy(pl.scan_csv(str(csv)), from_csv) + + parquet = home / "src.parquet" + pq.write_table(pa.table({"id": positions, "v": rng.normal(size=N)}), parquet, row_group_size=8192) + from_parquet = home / "from_parquet.parquet" + ordered_copy(pl.scan_parquet(str(parquet)), from_parquet) + + result = {"threads": pl.thread_pool_size(), "polars": pl.__version__} + for name, path in (("csv", from_csv), ("parquet", from_parquet)): + sizes, signature = layout(path) + table = pq.read_table(path) + result[name] = { + "row_groups": len(sizes), + "full_groups_exact": all(s == WRITE["row_group_size"] for s in sizes[:-1]), + "layout": signature, + "index_is_0_to_n": bool((table["__row_order"].to_numpy() == positions).all()), + "index_is_file_position": bool((table["__row_order"].to_numpy() == table["id"].to_numpy()).all()), + } + result["source_row_groups"] = pq.ParquetFile(parquet).metadata.num_row_groups + print(json.dumps(result)) + + +def main() -> None: + print(f"{N:,} rows; row groups of {WRITE['row_group_size']:,} asked for\n") + for threads in ("1", "3", None): + env = dict(os.environ) + if threads is None: + env.pop("POLARS_MAX_THREADS", None) + else: + env["POLARS_MAX_THREADS"] = threads + done = subprocess.run( + [sys.executable, __file__, "--child"], env=env, capture_output=True, text=True, check=True + ) + r = json.loads(done.stdout.strip().splitlines()[-1]) + groups = r["source_row_groups"] + print(f"polars {r['polars']} on {r['threads']} thread(s); the parquet source has {groups} row groups") + for name in ("csv", "parquet"): + s = r[name] + print( + f" {name:8s} source: {s['row_groups']} row groups, full groups exact={s['full_groups_exact']}, " + f"layout {s['layout']}, index 0..N-1={s['index_is_0_to_n']}, " + f"index equals file position={s['index_is_file_position']}" + ) + + +if __name__ == "__main__": + child() if "--child" in sys.argv else main() diff --git a/scripts/spike_reset_roundtrip.py b/scripts/spike_reset_roundtrip.py new file mode 100644 index 0000000..e9c1c4e --- /dev/null +++ b/scripts/spike_reset_roundtrip.py @@ -0,0 +1,93 @@ +"""ADR-007 evidence (D13, D14): a reset back and then forward, played through on today's code. + +Three kinds of file sit behind an entry, and ``reset_to`` treats them differently: + +- the entry directory: moved to the bullpen, copied back by a forward reset; +- the snapshots under ``compute_cache/``: the same, driven by a git-tracked list of the files that existed; +- the content-addressed clone of the source under ``data/.cas/``: DELETED by ``gc_cas``, not moved. + +Step s1 has an entry over ``orders.parquet``. Step s2 adds a cheap entry and a worthy entry over ``extra.parquet``. +The script resets to s1, resets forward to s2, and reads both s2 entries. ``extra.parquet`` is never touched. It then +empties the cache, which is the cold state of ADR-007 D7, and reads the worthy entry again so that it has to heal. + +Runs in a scratch ``TALLYMAN_HOME``. + + uv run python scripts/spike_reset_roundtrip.py +""" + +from __future__ import annotations + +import os +import tempfile +from pathlib import Path + +HOME = Path(tempfile.mkdtemp(prefix="spike_reset_roundtrip_")) +os.environ["TALLYMAN_HOME"] = str(HOME) +os.environ["XORQ_CACHE_DIR"] = str(HOME / "_global_xorq") # must be set before xorq is imported + +import pandas as pd # noqa: E402 + +from tallyman_core import catalog_state as cs # noqa: E402 +from tallyman_core import data_dir, ensure_project, set_active_project # noqa: E402 +from tallyman_core.paths import compute_cache_dir # noqa: E402 +from tallyman_xorq import build_and_persist # noqa: E402 +from tallyman_xorq.result_cache import cached_result_expr # noqa: E402 + +PROJECT = "spike" + + +def recipe(source: str, tail: str = "") -> str: + return ( + "from tallyman_xorq.io import read_project_file\n" + f"t = read_project_file({source!r}, project={PROJECT!r})\n" + f"expr = t{tail}\n" + ) + + +def clones() -> list[str]: + cas = data_dir(PROJECT) / ".cas" + return sorted(p.name[:8] for p in cas.iterdir()) if cas.is_dir() else [] + + +def read(label: str, content_hash: str) -> None: + cached_result_expr.cache_clear() + try: + print(f" {label}: ok, {len(cached_result_expr(PROJECT, content_hash).execute())} rows") + except Exception as exc: # noqa: BLE001 - the spike reports whatever is raised + print(f" {label}: raises {type(exc).__name__}: {str(exc)[:80]}") + + +def main() -> None: + ensure_project(PROJECT) + set_active_project(PROJECT) + cs.ensure_catalog_repo(PROJECT) + data = data_dir(PROJECT) + data.mkdir(parents=True, exist_ok=True) + pd.DataFrame({"region": ["a", "b", "a"], "price": [1.0, 2.0, 3.0]}).to_parquet(data / "orders.parquet") + pd.DataFrame({"k": ["x", "y", "x", "z"], "v": [1.0, 2.0, 3.0, 4.0]}).to_parquet(data / "extra.parquet") + + build_and_persist(PROJECT, recipe("orders.parquet")) + s1 = cs.checkpoint_catalog(PROJECT, "s1") + cheap = build_and_persist(PROJECT, recipe("extra.parquet", ".filter(t.v > 1)")) + worthy = build_and_persist(PROJECT, recipe("extra.parquet", ".group_by('k').aggregate(s=t.v.sum())")) + s2 = cs.checkpoint_catalog(PROJECT, "s2") + + print(f"at s2, clones: {clones()}") + read("cheap entry", cheap.content_hash) + read("worthy entry", worthy.content_hash) + + cs.reset_to(PROJECT, s1) + print(f"after the reset to s1, clones: {clones()}") + cs.reset_to(PROJECT, s2) + print(f"after the reset forward to s2, clones: {clones()}; extra.parquet is unchanged on disk") + read("cheap entry, which reads the clone on every read", cheap.content_hash) + read("worthy entry, whose snapshot came back from the bullpen", worthy.content_hash) + + for snapshot in compute_cache_dir(PROJECT).rglob("*.parquet"): + snapshot.unlink() + print("with the cache emptied, the worthy entry has to heal from its build:") + read("worthy entry", worthy.content_hash) + + +if __name__ == "__main__": + main() diff --git a/scripts/spike_row_order_joins.py b/scripts/spike_row_order_joins.py new file mode 100644 index 0000000..ab7600d --- /dev/null +++ b/scripts/spike_row_order_joins.py @@ -0,0 +1,93 @@ +"""ADR-008 evidence (D6): what happens to ``__row_order`` when entries are joined. + +Every file tallyman reads carries ``__row_order`` (ADR-008 D2), so both sides of a join have a column of that name and +ibis renames the right side's to ``__row_order_right``. Raw xorq plus tallyman's ``_canonical_sorted``. + +Questions, in the order printed: + +1. A two-way join: which columns come out? +2. A three-way join written in one recipe: does it build, and where does it fail? +3. A join ENTRY's snapshot keeps ``__row_order_right`` as data. Can that entry be joined to a third entry? +4. The same, when the writer drops ``__row_order_right`` from the snapshot (the rule D6 adopts). +5. What an author has to write for a three-way join in one recipe. +6. After a fan-out join or an outer join, is the left side's ``__row_order`` still unique and non-null? + + uv run python scripts/spike_row_order_joins.py +""" + +from __future__ import annotations + +import os +import tempfile +from pathlib import Path + +HOME = Path(tempfile.mkdtemp(prefix="spike_row_order_joins_")) +os.environ["XORQ_CACHE_DIR"] = str(HOME / "_global_xorq") # must be set before xorq is imported + +import pyarrow as pa # noqa: E402 +import pyarrow.parquet as pq # noqa: E402 +import xorq.api as xo # noqa: E402 +from xorq.ibis_yaml.compiler import build_expr # noqa: E402 + +from tallyman_xorq.source_cache import _canonical_sorted # noqa: E402 + +N = 5 + + +def entry(name: str) -> Path: + path = HOME / f"{name}.parquet" + pq.write_table( + pa.table({"k": list(range(N)), name: [f"{name}{i}" for i in range(N)], "__row_order": list(range(N))}), path + ) + return path + + +def attempt(label: str, fn) -> object: + try: + out = fn() + except Exception as exc: # noqa: BLE001 - the spike reports whatever is raised + print(f" {label}: raises {type(exc).__name__}: {str(exc)[:100]}") + return None + shown = list(out.columns) if hasattr(out, "columns") else out + print(f" {label}: ok -> {shown}") + return out + + +def main() -> None: + a, b, c = (xo.deferred_read_parquet(str(entry(n))) for n in ("a", "b", "c")) + + print("1. two-way join") + ab = attempt("a.join(b, 'k')", lambda: a.join(b, "k")) + + print("2. three-way join in one recipe") + abc = a.join(b, "k").join(c, "k") + attempt("the expression's columns", lambda: abc) + attempt("build_expr", lambda: Path(build_expr(abc, builds_dir=HOME / "builds")).name) + attempt("_canonical_sorted (the next build step)", lambda: _canonical_sorted(abc)) + attempt("execute", lambda: len(abc.execute())) + + print("3. a join entry's snapshot, with __row_order_right kept as data, joined to a third entry") + kept = HOME / "ab_kept.parquet" + pq.write_table(ab.to_pyarrow(), kept) + ab_kept = xo.deferred_read_parquet(str(kept)) + attempt("execute", lambda: len(ab_kept.join(c, "k").execute())) + + print("4. the same, when the writer drops __row_order_right") + dropped = HOME / "ab_dropped.parquet" + pq.write_table(ab.to_pyarrow().drop_columns(["__row_order_right"]), dropped) + ab_dropped = xo.deferred_read_parquet(str(dropped)) + attempt("execute", lambda: len(ab_dropped.join(c, "k").execute())) + + print("5. a three-way join in one recipe, with __row_order dropped on the right-hand inputs") + attempt("execute", lambda: len(a.join(b.drop("__row_order"), "k").join(c.drop("__row_order"), "k").execute())) + + print("6. the left side's __row_order after a fan-out join and after an outer join") + twice = xo.union(b, b).drop("__row_order") + fan = a.join(twice, "k").execute() + print(f" fan-out join: {len(fan)} rows, {fan['__row_order'].nunique()} distinct values of __row_order") + outer = a.filter(a.k < 3).outer_join(c.drop("__row_order"), "k").execute() + print(f" outer join: {len(outer)} rows, {int(outer['__row_order'].isna().sum())} with a null __row_order") + + +if __name__ == "__main__": + main() diff --git a/scripts/spike_single_partition_loaded_build.py b/scripts/spike_single_partition_loaded_build.py new file mode 100644 index 0000000..be688c6 --- /dev/null +++ b/scripts/spike_single_partition_loaded_build.py @@ -0,0 +1,98 @@ +"""ADR-009 evidence (D1): making a LOADED BUILD run single-partition. + +``scripts/spike_float_aggregate_digest.py`` runs SQL on a connection it made and configured itself. ``materialize`` +(ADR-007 D4) executes a build that ``load_expr`` loaded, and ``load_expr`` makes its own backend objects. + +Questions, in the order printed. A float SUM and AVG group-by over 3,000,000 rows, distinct value digests in 5 runs: + +1. The loaded build, executed as loaded. +2. The same, while a single-partition connection exists on the side. Which backend does the build run on? +3. The loaded build rebound onto the single-partition connection with ``replace_sources``. +4. ``SET`` applied to the backends the load made, with no rebinding. +5. Does either change the process default backend, which serves page reads? + + uv run python scripts/spike_single_partition_loaded_build.py +""" + +from __future__ import annotations + +import hashlib +import os +import tempfile +from pathlib import Path + +HOME = Path(tempfile.mkdtemp(prefix="spike_single_partition_")) +os.environ["XORQ_CACHE_DIR"] = str(HOME / "_global_xorq") # must be set before xorq is imported + +import numpy as np # noqa: E402 +import pyarrow as pa # noqa: E402 +import pyarrow.parquet as pq # noqa: E402 +import xorq.api as xo # noqa: E402 +from xorq.common.utils.graph_utils import find_all_sources, replace_sources # noqa: E402 +from xorq.config import default_backend # noqa: E402 +from xorq.ibis_yaml.compiler import build_expr, load_expr # noqa: E402 + +N = 3_000_000 +RUNS = 5 +SINGLE = "SET datafusion.execution.target_partitions = 1" + + +def digest(expr) -> str: + table = expr.to_pyarrow() + h = hashlib.sha256() + for column in ("s", "m"): + h.update(np.asarray(table[column].to_numpy()).tobytes()) + return h.hexdigest()[:10] + + +def partitions(con) -> str: + return str(con.raw_sql("SHOW datafusion.execution.target_partitions").to_pandas().iloc[0, 1]) + + +def report(label: str, make) -> None: + seen = {digest(make()) for _ in range(RUNS)} + print(f" {label}: {len(seen)} distinct digest(s) in {RUNS} runs") + + +def main() -> None: + rng = np.random.default_rng(3) + source = HOME / "source.parquet" + table = pa.table({"g": rng.integers(0, 500, size=N), "v": rng.normal(scale=1e6, size=N)}) + pq.write_table(table, source, row_group_size=100_000) + t = xo.deferred_read_parquet(str(source)) + build_dir = Path( + build_expr(t.group_by("g").agg(s=t.v.sum(), m=t.v.mean()).order_by("g"), builds_dir=HOME / "builds") + ) + + print("1. the loaded build, executed as loaded") + report("as loaded", lambda: load_expr(build_dir)) + + print("2. a single-partition connection on the side; the build is not rebound") + side = xo.connect() + side.raw_sql(SINGLE) + backends = find_all_sources(load_expr(build_dir)) + print(f" side connection reports {partitions(side)}; the build's {len(backends)} backend(s) report", end=" ") + print( + f"{[partitions(b) for b in backends]}, and none is the side connection: {all(b is not side for b in backends)}" + ) + report("not rebound", lambda: load_expr(build_dir)) + + def rebound(): + loaded = load_expr(build_dir) + return replace_sources({id(b): side for b in find_all_sources(loaded)}, loaded) + + def set_on_loaded(): + loaded = load_expr(build_dir) + for b in find_all_sources(loaded): + b.raw_sql(SINGLE) + return loaded + + print("3. rebound onto the single-partition connection") + report("replace_sources", rebound) + print("4. SET on the backends the load made") + report("SET on each", set_on_loaded) + print(f"5. the process default backend reports {partitions(default_backend())}") + + +if __name__ == "__main__": + main() diff --git a/scripts/spike_sort_grafting.py b/scripts/spike_sort_grafting.py new file mode 100644 index 0000000..1f3c393 --- /dev/null +++ b/scripts/spike_sort_grafting.py @@ -0,0 +1,139 @@ +"""ADR-008 evidence (D10, D11, and the uniqueness note under D5): where a unique sort has to be imposed. + +Paddy's rule: when a supplied sort is not deterministic, the natural order is imposed into each ``order_by``. Today +tallyman extends an author's sort only when it is the TOP node of the expression (``source_cache._canonical_sorted``). +``original_row_order`` and ``id`` stand in for ``__row_order``, which is not implemented yet. + +Questions, in the order printed: + +1. An ``order_by`` followed by another step: in what order is the snapshot written? +2. A top-3 entry: is it written in rank order? +3. A sort that feeds a ``limit``, with ties at the cut: which ROWS come back, run to run? +4. A window function: are its VALUES the same run to run, with no order, a tied order, and a unique order? +5. Can parquet statistics tell a unique column from one with duplicates? + + uv run python scripts/spike_sort_grafting.py +""" + +from __future__ import annotations + +import hashlib +import os +import tempfile +from pathlib import Path + +HOME = Path(tempfile.mkdtemp(prefix="spike_sort_grafting_")) +os.environ["XORQ_CACHE_DIR"] = str(HOME / "_global_xorq") # must be set before xorq is imported + +import numpy as np # noqa: E402 +import pyarrow as pa # noqa: E402 +import pyarrow.parquet as pq # noqa: E402 +import xorq.api as xo # noqa: E402 +import xorq.vendor.ibis.expr.operations as ops # noqa: E402 + +from tallyman_xorq.source_cache import _canonical_sorted, _is_worthy_expr # noqa: E402 + +N = 3_000_000 +RUNS = 5 + + +def connection(partitions: int | None): + con = xo.connect() + if partitions is not None: + con.raw_sql(f"SET datafusion.execution.target_partitions = {partitions}") + return con + + +def keys_of(expr) -> list[str]: + node = expr.op() + if not isinstance(node, ops.Sort): + return [] + return [k.expr.name + ("" if k.ascending else " desc") for k in node.keys] + + +def nonfinal_sorts() -> None: + small = HOME / "small.parquet" + rows = {"name": list("abcdef"), "amount": [40, 10, 60, 20, 50, 30], "original_row_order": list(range(6))} + pq.write_table(pa.table(rows), small) + t = xo.deferred_read_parquet(str(small)) + by_amount = t.order_by(t.amount.desc()) + shapes = { + "order_by last": by_amount, + "order_by, then mutate": by_amount.mutate(double=t.amount * 2), + "order_by, then select": by_amount.select("name", "amount", "original_row_order"), + "order_by, then filter": by_amount.filter(t.amount > 15), + } + print("1. parent rows are in the order 40, 10, 60, 20, 50, 30; the author asks for amount descending") + for label, expr in shapes.items(): + written = _canonical_sorted(expr) + print(f" {label:24s} worthy={_is_worthy_expr(expr)!s:5s} sort keys={keys_of(written)}") + print(f" {'':24s} written as {written.execute()['amount'].tolist()}") + top3 = _canonical_sorted(by_amount.limit(3)).execute()["amount"].tolist() + print(f"2. top 3 by amount is written as {top3}; the author asked for [60, 50, 40]") + + +def big_file() -> Path: + rng = np.random.default_rng(5) + path = HOME / "big.parquet" + table = pa.table({"id": np.arange(N), "g": rng.integers(0, 200, size=N), "v": rng.integers(0, 1000, size=N)}) + pq.write_table(table, path, row_group_size=100_000) + return path + + +def sort_feeds_limit(path: Path) -> None: + def row_set(partitions, keys) -> str: + t = xo.deferred_read_parquet(str(path), con=connection(partitions)) + ids = np.sort(t.order_by(keys).limit(1000).to_pyarrow()["id"].to_numpy()) + return hashlib.md5(ids.tobytes()).hexdigest()[:10] + + print(f"3. order_by(g).limit(1000) over {N:,} rows; about 15,000 rows tie on the smallest g") + unique_answer = row_set(1, ["g", "id"]) + for label, partitions, keys in ( + ("default connection, key g", None, ["g"]), + ("single-partition, key g", 1, ["g"]), + ("default connection, key (g, id)", None, ["g", "id"]), + ): + seen = [row_set(partitions, keys) for _ in range(RUNS)] + same = all(s == unique_answer for s in seen) + print(f" {label:34s} {len(set(seen))} distinct row set(s) in {RUNS} runs; equals the (g, id) answer: {same}") + + +def window_values(path: Path) -> None: + def digest(partitions, shape) -> str: + t = xo.deferred_read_parquet(str(path), con=connection(partitions)) + total = { + "no order": lambda: t.v.cumsum(), + "order_by g (tied)": lambda: t.v.cumsum(order_by=t.g), + "order_by (g, id)": lambda: t.v.cumsum(order_by=[t.g, t.id]), + }[shape]() + out = t.mutate(c=total).order_by("id").select("c").to_pyarrow() # final order fixed, so only values can differ + return hashlib.md5(out["c"].to_numpy().tobytes()).hexdigest()[:10] + + print("4. cumsum() as a computed column; distinct sets of VALUES") + for shape in ("no order", "order_by g (tied)", "order_by (g, id)"): + for label, partitions in (("default connection", None), ("single-partition", 1)): + seen = {digest(partitions, shape) for _ in range(RUNS)} + print(f" {shape:20s} {label:20s} {len(seen)} in {RUNS} runs") + + +def statistics() -> None: + path = HOME / "stats.parquet" + pq.write_table(pa.table({"unique": [0, 1, 2, 3, 4], "duplicated": [0, 0, 2, 3, 4]}), path, write_statistics=True) + group = pq.ParquetFile(path).metadata.row_group(0) + print("5. parquet statistics of a unique column and of one holding 0 twice") + for i in range(2): + s = group.column(i).statistics + name = group.column(i).path_in_schema + print(f" {name:11s} min={s.min} max={s.max} nulls={s.null_count} has_distinct_count={s.has_distinct_count}") + + +def main() -> None: + nonfinal_sorts() + path = big_file() + sort_feeds_limit(path) + window_values(path) + statistics() + + +if __name__ == "__main__": + main() diff --git a/scripts/spike_stream_order.py b/scripts/spike_stream_order.py new file mode 100644 index 0000000..0eed4bd --- /dev/null +++ b/scripts/spike_stream_order.py @@ -0,0 +1,82 @@ +"""ADR-007 evidence (open question 1, the alternative that was not taken) and ADR-009 (D1). + +On the connection ``materialize`` uses (``target_partitions = 1``), do the rows of a row-preserving plan reach the +writer in the parent file's order? If so, a writer could number rows in parent order with no ``__row_order`` column +carried through the recipe, which is what materializing every entry would have relied on. It is also why a bare +``limit`` or a window function with no order is repeatable at materialization: by the engine's behaviour, not by the +query's own meaning (ADR-008 D10). + +The parent is 3,000,000 rows in 100,000-row groups, above the 10,485,760-byte scan-split threshold. ``id`` is the file +position. Each plan is streamed three times per connection. + + uv run python scripts/spike_stream_order.py +""" + +from __future__ import annotations + +import os +import tempfile +from pathlib import Path + +HOME = Path(tempfile.mkdtemp(prefix="spike_stream_order_")) +os.environ["XORQ_CACHE_DIR"] = str(HOME / "_global_xorq") # must be set before xorq is imported + +import numpy as np # noqa: E402 +import pyarrow as pa # noqa: E402 +import pyarrow.parquet as pq # noqa: E402 +import xorq.api as xo # noqa: E402 + +N = 3_000_000 +RUNS = 3 + + +def in_file_order(expr) -> bool: + last = -1 + for batch in expr.to_pyarrow_batches(): + ids = batch.column("id").to_numpy() + if len(ids) == 0: + continue + if ids[0] < last or (np.diff(ids) < 0).any(): + return False + last = ids[-1] + return True + + +def plans(t) -> dict: + return { + "bare read": t, + "filter + computed column": t.filter(t.g < 150).mutate(w=t.v * 2), + "select two columns": t.select("id", "s"), + "string filter + cast": t.filter(t.s.endswith("7")).mutate(gf=t.g.cast("float64")), + "window function with no order": t.mutate(c=t.v.cumsum()), + } + + +def main() -> None: + rng = np.random.default_rng(11) + parent = HOME / "parent.parquet" + columns = { + "id": np.arange(N), + "g": rng.integers(0, 200, size=N), + "v": rng.normal(size=N), + "s": pa.array([f"row-{i % 1000}" for i in range(N)]), + } + pq.write_table(pa.table(columns), parent, row_group_size=100_000) + print( + f"parent: {parent.stat().st_size / 1e6:.0f} MB, {pq.ParquetFile(parent).metadata.num_row_groups} row groups\n" + ) + + for label, partitions in (("target_partitions = 1", 1), ("default connection", None)): + print(label) + for name in plans(xo.deferred_read_parquet(str(parent))): + results = [] + for _ in range(RUNS): + con = xo.connect() + if partitions is not None: + con.raw_sql(f"SET datafusion.execution.target_partitions = {partitions}") + results.append(in_file_order(plans(xo.deferred_read_parquet(str(parent), con=con))[name])) + print(f" {name:32s} streamed in file order: {results}") + + +if __name__ == "__main__": + main() From f9bb94bc3c85a388289718a2db5a22f756905f30 Mon Sep 17 00:00:00 2001 From: Paddy Mullen Date: Mon, 21 Sep 2026 00:14:18 -0400 Subject: [PATCH 005/111] test(cache): failing tests for ADR-007, ADR-008 and ADR-009 One commit of failing tests for the whole cache redesign, so every test is seen red on CI before the change lands (ADR-007 D9, step 1): - ADR-007: builds carry no cache nodes, snapshot path from the content hash, chaining as a bare read of the parent's snapshot, one writer (materialize), ensure_materialized, re-creation of each class of file, the Buckaroo hand-off (derived session ids, view build, forced reload, klass reload), one write at a time per project, reset leaves compute_cache alone, the xorq-cache sentinel. - ADR-008: __row_order on every file, the cheap/worthy allow-list, build errors for a dropped or assigned column, joins, sort grafting and hoisting, repeatable pages, CSV roots and #168, raw parquet reads. - ADR-009: the content digest, the pinned snapshot format, the single-partition connection, create runs the query twice, engine versions. Co-Authored-By: Claude Sonnet 5 --- tests/big_parquet.py | 37 ++ tests/test_buckaroo_handoff.py | 383 +++++++++++++ tests/test_cache_files.py | 216 ++++++++ tests/test_digest.py | 281 ++++++++++ tests/test_materialize.py | 546 +++++++++++++++++++ tests/test_project_lock.py | 262 +++++++++ tests/test_reset_keeps_compute_cache.py | 159 ++++++ tests/test_row_order.py | 527 ++++++++++++++++++ tests/test_row_order_pages.py | 91 ++++ tests/test_row_order_sorts.py | 160 ++++++ tests/test_snapshot_format.py | 691 ++++++++++++++++++++++++ tests/test_tallyman_read_csv.py | 416 ++++++-------- 12 files changed, 3529 insertions(+), 240 deletions(-) create mode 100644 tests/big_parquet.py create mode 100644 tests/test_buckaroo_handoff.py create mode 100644 tests/test_cache_files.py create mode 100644 tests/test_digest.py create mode 100644 tests/test_materialize.py create mode 100644 tests/test_project_lock.py create mode 100644 tests/test_reset_keeps_compute_cache.py create mode 100644 tests/test_row_order.py create mode 100644 tests/test_row_order_pages.py create mode 100644 tests/test_row_order_sorts.py create mode 100644 tests/test_snapshot_format.py diff --git a/tests/big_parquet.py b/tests/big_parquet.py new file mode 100644 index 0000000..214e2d7 --- /dev/null +++ b/tests/big_parquet.py @@ -0,0 +1,37 @@ +"""A parquet fixture above DataFusion's scan-split threshold, shared by the cache-redesign tests. + +DataFusion splits a file scan into parallel byte ranges when the file is larger than +``datafusion.optimizer.repartition_file_min_size`` (10,485,760 bytes in xorq-datafusion 0.2.7), and only then does an +unsorted ``LIMIT/OFFSET`` return rows in an unstable order (ADR-008) or a float aggregate merge partial sums in an +unstable order (ADR-009). A fixture below that size proves nothing about either, so the writer asserts its size. +""" + +from __future__ import annotations + +from pathlib import Path + +SPLIT_THRESHOLD_BYTES = 10_485_760 + + +def write_big_parquet(path: Path, n_rows: int = 1_500_000, seed: int = 0, row_group_size: int = 100_000) -> Path: + """Write ``n_rows`` rows: ``id`` (the file position), ``g`` (200 values, so a sort on it has ties), ``v`` (float). + + ``v`` is random and therefore incompressible, which is what makes the file larger than the split threshold. + """ + import numpy as np + import pyarrow as pa + import pyarrow.parquet as pq + + rng = np.random.default_rng(seed) + table = pa.table( + { + "id": np.arange(n_rows, dtype=np.int64), + "g": rng.integers(0, 200, size=n_rows, dtype=np.int64), + "v": rng.normal(scale=1e6, size=n_rows), + } + ) + path.parent.mkdir(parents=True, exist_ok=True) + pq.write_table(table, path, row_group_size=row_group_size) + size = path.stat().st_size + assert size > SPLIT_THRESHOLD_BYTES, f"fixture is {size} bytes, below the {SPLIT_THRESHOLD_BYTES}-byte threshold" + return path diff --git a/tests/test_buckaroo_handoff.py b/tests/test_buckaroo_handoff.py new file mode 100644 index 0000000..32f57c2 --- /dev/null +++ b/tests/test_buckaroo_handoff.py @@ -0,0 +1,383 @@ +"""The Buckaroo hand-off: tallyman gives Buckaroo files that already exist, and remembers no sessions. + +Red tests for ``plans/ADR-007-tallyman-owned-materialization.md`` D6 (Buckaroo is handed something that already exists) +and D12 (files are deleted only by an explicit user action), plus the payload hint of +``plans/ADR-008-row-order-of-reads.md`` D8 (Buckaroo's half of the row-order contract). + +Buckaroo is faked with an ``httpx.MockTransport``. The fake follows the parts of ``buckaroo/server/handlers.py`` +(0.15.6) that tallyman relies on: the session id comes from the POST body, and a ``/load_expr`` is a warm hit (the +pipeline is not re-run) only when that id already exists with the same ``build_dir``, ``force_reload`` is not set, and +the body carries none of ``component_config``, ``column_config_overrides``, ``extra_grid_config``, ``init_sd`` and +``skip_stat_columns``. ``/reload_expr/`` answers 404 for a session Buckaroo does not have. +""" + +from __future__ import annotations + +import json +import shutil +import uuid +from pathlib import Path + +import httpx +import pytest +from fastapi.testclient import TestClient + +from tallyman_companion import create_app +from tallyman_companion.buckaroo_lifecycle import BuckarooManager +from tallyman_core import ( + data_dir, + entry_dir, + entry_expanded_build_dir, + entry_manifest_path, + entry_stat_cache_dir, + read_manifest, +) +from tallyman_core.paths import compute_cache_dir, tallyman_home +from tallyman_xorq import build_and_persist + +_CONFIG_KEYS = ("component_config", "column_config_overrides", "extra_grid_config", "init_sd", "skip_stat_columns") + + +def _agg_code(project: str) -> str: # an Aggregate: a worthy entry, materialized to a snapshot when it is created + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.group_by("region").aggregate(total=t.price.sum(), n=t.count()) +""" + + +def _second_agg_code(project: str) -> str: # a different worthy entry in the same project + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.group_by("category").aggregate(n=t.count()) +""" + + +def _cheap_code(project: str) -> str: # a filter over one file: a cheap entry, re-run on every read + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.filter(t.price > 0) +""" + + +def _sid(project: str, content_hash: str) -> str: + """The session id ADR-007 D6 derives from the project and the content hash.""" + return f"entry-{project}-{content_hash}" + + +# Names this change introduces are imported inside one-line helpers. Imported at module level, or next to other imports +# in one block, they would be classified by ruff's isort as third-party until the module exists, and the lint job that +# gates the test jobs on CI would fail before any test ran. + + +def _snapshot_path(project: str, content_hash: str) -> Path: + from tallyman_xorq.materialize import snapshot_path + + return snapshot_path(project, content_hash) + + +def _view_build_dir(project: str, content_hash: str) -> Path: + from tallyman_core.paths import entry_view_build_dir + + return entry_view_build_dir(project, content_hash) + + +class _AliveProc: + def poll(self): + return None # looks alive, so BuckarooManager.is_running is True + + +class FakeBuckaroo: + """The slice of Buckaroo's HTTP API that tallyman uses, keeping its own sessions and a log of every request.""" + + def __init__(self) -> None: + self.sessions: dict[str, dict] = {} + self.requests: list[dict] = [] + self.warm_hits = 0 + # Called with the parsed body when a /load_expr arrives, before it is answered: lets a test look at the + # disk at the moment "Buckaroo is called". + self.on_load = None + + def loads(self) -> list[dict]: + return [r["body"] for r in self.requests if r["path"] == "/load_expr"] + + def reloads(self) -> list[str]: + return [r["path"] for r in self.requests if r["path"].startswith("/reload_expr/")] + + def handler(self, request: httpx.Request) -> httpx.Response: + path = request.url.path + body = json.loads(request.content) if request.content else {} + self.requests.append({"method": request.method, "path": path, "body": body}) + if path == "/load_expr": + if self.on_load is not None: + self.on_load(body) + session = body.get("session") or uuid.uuid4().hex + existing = self.sessions.get(session) + has_config = any(body.get(k) for k in _CONFIG_KEYS) + same_build = bool(existing) and existing["build_dir"] == body["build_dir"] + if same_build and not body.get("force_reload") and not has_config: + self.warm_hits += 1 + return httpx.Response(200, json={"session": session}) + self.sessions[session] = {"build_dir": body["build_dir"]} + return httpx.Response(200, json={"session": session}) + if path.startswith("/reload_expr/"): + if path.rsplit("/", 1)[1] not in self.sessions: + return httpx.Response(404, json={"error": "unknown session"}) + return httpx.Response(200, json={}) + return httpx.Response(404, json={"error": f"unhandled {path}"}) + + +def _manager(fake: FakeBuckaroo) -> BuckarooManager: + bk = BuckarooManager() + bk.proc = _AliveProc() + bk.bound_port = 8799 + bk._maybe_restart = lambda: None + bk._client = httpx.Client(transport=httpx.MockTransport(fake.handler)) + return bk + + +# --------------------------------------------------------------------------- +# sessions are derived, never remembered (ADR-007 D6) +# --------------------------------------------------------------------------- + + +def test_a_session_buckaroo_forgot_is_posted_again(project, orders_parquet): + """ADR-007 D6 (Buckaroo is handed something that already exists): Buckaroo drops a session that has had no + browser attached for an hour, and tallyman used to answer the next open from its own session map, handing the grid + an id Buckaroo no longer knew. Now every open posts ``/load_expr`` with the derived id; Buckaroo's own warm-hit + rule makes the repeat cheap when it still has the session.""" + h = build_and_persist(project, _agg_code(project)).content_hash + fake = FakeBuckaroo() + bk = _manager(fake) + + assert bk.load_session(h, project)["status"] == "ok" + assert len(fake.loads()) == 1 + + fake.sessions.clear() # Buckaroo evicts the idle session + result = bk.load_session(h, project) + + assert result["status"] == "ok" + assert len(fake.loads()) == 2, "tallyman answered from its own session map instead of posting to Buckaroo" + assert fake.loads()[1]["session"] == _sid(project, h) + assert _sid(project, h) in fake.sessions + assert result["session_id"] == _sid(project, h) + + +def test_a_repeat_open_of_an_ordinary_entry_is_a_warm_hit_in_buckaroo(project, orders_parquet): + """ADR-007 D6: an ordinary entry posts none of the config-bearing fields and the same ``build_dir`` every time, so + the second post is a no-op inside Buckaroo. This is what replaces tallyman's session map.""" + h = build_and_persist(project, _agg_code(project)).content_hash + fake = FakeBuckaroo() + bk = _manager(fake) + + assert bk.load_session(h, project)["status"] == "ok" + assert bk.load_session(h, project)["status"] == "ok" + + assert len(fake.loads()) == 2, "the second open never reached Buckaroo" + assert fake.warm_hits == 1 + + +def test_session_id_is_derived_from_the_project_and_the_hash(isolated_home): + """ADR-007 D6: the id is a function of the project and the content hash. Putting the project in it also closes + #172, one project's session served to another on a hash collision.""" + bk = BuckarooManager() + + assert bk.session_id_for("proj", "abc123") == "entry-proj-abc123" + assert bk.session_id_for("one", "abc123") != bk.session_id_for("two", "abc123") + + +def test_tallyman_keeps_no_record_of_buckaroo_sessions(project, orders_parquet): + """ADR-007 D6: ``_sessions``, ``evict_session``, ``~/.tallyman/buckaroo_sessions.json`` and the session count are + retired, because nothing needs to know which grids are open.""" + h = build_and_persist(project, _agg_code(project)).content_hash + bk = _manager(FakeBuckaroo()) + + assert bk.load_session(h, project)["status"] == "ok" + + assert not (tallyman_home() / "buckaroo_sessions.json").exists(), "a session file was written" + assert not hasattr(bk, "_sessions") + assert not hasattr(bk, "evict_session") + assert "session_count" not in bk.status() + + +# --------------------------------------------------------------------------- +# what Buckaroo is handed (ADR-007 D6) +# --------------------------------------------------------------------------- + + +def test_worthy_entry_is_handed_a_view_build_of_its_snapshot(project, orders_parquet): + """ADR-007 D6: a worthy entry's grid is posted a *view build*: a build whose whole graph is one bare read of the + entry's snapshot. Buckaroo then never runs the aggregate, join or sort, and never writes a snapshot.""" + import xorq.vendor.ibis.expr.operations as ops + from xorq.common.utils.graph_utils import walk_nodes + from xorq.expr.relations import Read + from xorq.ibis_yaml.compiler import load_expr + from xorq.vendor.ibis.expr.operations.core import Node + + h = build_and_persist(project, _agg_code(project)).content_hash + fake = FakeBuckaroo() + bk = _manager(fake) + assert bk.load_session(h, project)["status"] == "ok" + + posted = Path(fake.loads()[0]["build_dir"]) + loaded = load_expr(posted) + relations = sorted({type(n).__name__ for n in walk_nodes((Node,), loaded) if isinstance(n, ops.Relation)}) + assert relations == ["Read"], f"the posted build does more than read a file: {relations}" + + view_dir = _view_build_dir(project, h) + assert posted == view_dir or view_dir in posted.parents, f"{posted} is not under {view_dir}" + reads = list(walk_nodes(Read, loaded)) + assert len(reads) == 1 + assert Path(dict(reads[0].read_kwargs)["hash_path"]).resolve() == _snapshot_path(project, h).resolve() + + +def test_view_build_directory_is_stable_across_opens(project, orders_parquet): + """ADR-007 D6: Buckaroo's stat-cache keys include the build directory's path, so the view build is written once + to a stable per-entry directory and the same path is posted every time.""" + h = build_and_persist(project, _agg_code(project)).content_hash + fake = FakeBuckaroo() + bk = _manager(fake) + + assert bk.load_session(h, project)["status"] == "ok" + fake.sessions.clear() + assert bk.load_session(h, project)["status"] == "ok" + + bodies = fake.loads() + assert len(bodies) == 2, "the second open never reached Buckaroo" + assert bodies[0]["build_dir"] == bodies[1]["build_dir"] + + +def test_cheap_entry_is_handed_its_own_expanded_build_over_files_that_exist(project, orders_parquet): + """ADR-007 D6 and D13: a cheap entry is a stored plan over files that exist. ``load_session`` makes them exist + first (``ensure_materialized``), so with the clone and every cached copy of the source deleted, the build posted + to Buckaroo still reads files that are there at the moment it is posted.""" + from xorq.common.utils.graph_utils import walk_nodes + from xorq.expr.relations import Read + from xorq.ibis_yaml.compiler import load_expr + + h = build_and_persist(project, _cheap_code(project)).content_hash + shutil.rmtree(data_dir(project) / ".cas", ignore_errors=True) + shutil.rmtree(compute_cache_dir(project), ignore_errors=True) + + reads_at_post: list[Path] = [] + + def at_post(body: dict) -> None: + loaded = load_expr(Path(body["build_dir"])) + reads_at_post.extend(Path(dict(r.read_kwargs)["hash_path"]) for r in walk_nodes(Read, loaded)) + + fake = FakeBuckaroo() + fake.on_load = at_post + bk = _manager(fake) + assert bk.load_session(h, project)["status"] == "ok" + + assert fake.loads()[0]["build_dir"] == str(entry_expanded_build_dir(project, h)) + assert reads_at_post, "the posted build reads no file at all" + missing = [p for p in reads_at_post if not p.exists()] + assert not missing, f"Buckaroo was handed a build that reads files which do not exist: {missing}" + + +@pytest.mark.parametrize("recipe", [_agg_code, _cheap_code], ids=["worthy", "cheap"]) +def test_every_load_expr_body_names_the_row_order_column(project, orders_parquet, recipe): + """ADR-008 D8 (Buckaroo's half is one hint): every ``/load_expr`` payload carries ``row_order_column`` so that + Buckaroo can order and page by it (buckaroo-data/buckaroo#974). An ordinary entry sends none of the fields that + would defeat Buckaroo's warm-hit short-circuit.""" + h = build_and_persist(project, recipe(project)).content_hash + fake = FakeBuckaroo() + bk = _manager(fake) + + assert bk.load_session(h, project)["status"] == "ok" + + body = fake.loads()[0] + assert body.get("row_order_column") == "__row_order" + assert body.get("session") == _sid(project, h) + assert not any(body.get(k) for k in _CONFIG_KEYS) + + +# --------------------------------------------------------------------------- +# a deleted or unfaithful file is put right before Buckaroo is called (ADR-007 D5, D6, D12) +# --------------------------------------------------------------------------- + + +def test_explicit_delete_then_open_heals_and_verifies_before_buckaroo_is_called(project, orders_parquet): + """ADR-007 D12 (files are deleted only by an explicit user action) and D5 (``ensure_materialized``): the Cache + page's delete removes the snapshot, and the next open rewrites it and checks it against the recorded digest + before the grid is posted. Buckaroo never has to repair a file.""" + from tallyman_xorq.result_cache import baked_snapshot_path, snapshot_file_digest + + h = build_and_persist(project, _agg_code(project)).content_hash + snap = baked_snapshot_path(project, h) + assert snap is not None and snap.exists() + client = TestClient(create_app(project)) + assert client.delete(f"/{project}/api/result_cache/{h}").status_code == 200 + assert not snap.exists() + + seen: dict = {} + + def at_post(body: dict) -> None: + seen["exists"] = snap.exists() + seen["digest"] = snapshot_file_digest(snap) if snap.exists() else None + + fake = FakeBuckaroo() + fake.on_load = at_post + bk = _manager(fake) + assert bk.load_session(h, project)["status"] == "ok" + + assert seen.get("exists"), "Buckaroo was called before the deleted snapshot was rewritten" + assert seen["digest"] == read_manifest(entry_dir(project, h)).result_digest + + +def test_unfaithful_heal_forces_a_reload_of_the_open_grid(project, orders_parquet): + """ADR-007 D6: an unfaithful heal is the one event that leaves an open session wrong, since the path now holds + other rows than the stats Buckaroo computed. tallyman keeps no session record to drop, so the companion's hook + wipes the entry's stat cache and posts ``/load_expr`` for the derived id with ``force_reload`` set.""" + from tallyman_xorq.result_cache import baked_snapshot_path, cached_result_expr + + h = build_and_persist(project, _agg_code(project)).content_hash + fake = FakeBuckaroo() + bk = _manager(fake) + with TestClient(create_app(project, buckaroo=bk)): # entering runs the startup handlers that register the hook + assert bk.load_session(h, project)["status"] == "ok" # the grid is open + + stale = entry_stat_cache_dir(project, h) / "parquet" / "stale-stats.parquet" + stale.parent.mkdir(parents=True, exist_ok=True) + stale.write_text("stale") + manifest_path = entry_manifest_path(project, h) + doc = json.loads(manifest_path.read_text()) + doc["result_digest"] = "arrow-sha256:" + "0" * 64 # no rewrite can match this + manifest_path.write_text(json.dumps(doc)) + snap = baked_snapshot_path(project, h) + assert snap is not None + snap.unlink() + cached_result_expr.cache_clear() + + seen: list[tuple] = [] + fake.on_load = lambda body: seen.append((body.get("force_reload"), stale.exists())) + cached_result_expr(project, h) # the heal; its digest cannot match the recorded one + + forced = [b for b in fake.loads() if b.get("force_reload")] + assert len(forced) == 1, "the unfaithful-heal hook did not post a forced reload to Buckaroo" + assert forced[0]["session"] == _sid(project, h) + assert forced[0].get("row_order_column") == "__row_order" + assert forced[0].get("no_browser") is True + assert Path(forced[0]["build_dir"]).is_dir() + assert seen == [(True, False)], "the reload must be posted after the stale stats are wiped" + + +def test_klass_reload_reaches_every_entry_without_a_session_record(project, orders_parquet): + """ADR-007 D6: ``reload_project_sessions`` used to walk tallyman's own record of open sessions, which the design + deletes, and Buckaroo has no route that lists sessions. With derived ids it posts ``/reload_expr/`` for every + entry of the project and treats Buckaroo's 404 as "not open". An entry that was never opened gets no session.""" + a = build_and_persist(project, _agg_code(project)).content_hash + b = build_and_persist(project, _second_agg_code(project)).content_hash + fake = FakeBuckaroo() + bk = _manager(fake) + assert bk.load_session(a, project)["status"] == "ok" # A is open in a tab; B never was + + reloaded = bk.reload_project_sessions(project) + + assert sorted(fake.reloads()) == sorted([f"/reload_expr/{_sid(project, a)}", f"/reload_expr/{_sid(project, b)}"]) + assert reloaded == 1, "only the open grid can have been reloaded" + assert _sid(project, b) not in fake.sessions + assert not hasattr(bk, "_sessions") diff --git a/tests/test_cache_files.py b/tests/test_cache_files.py new file mode 100644 index 0000000..6160568 --- /dev/null +++ b/tests/test_cache_files.py @@ -0,0 +1,216 @@ +"""Files are written only because something is about to read them, and deleted only by an explicit user action. + +Red tests for ``plans/ADR-007-tallyman-owned-materialization.md`` D12 (files are deleted only by an explicit user +action), D5 (the verify sweep is not a caller of ``ensure_materialized``) and D14 (the Cache page needs a row for a +file whose entry is not in the catalog), and for ``plans/ADR-009-digest-stability.md`` D6 (an entry whose recipe is not +reproducible has its file pinned, and the Cache page's delete skips it and says why). + +The Cache page is ``GET /{project}/api/result_cache`` and ``DELETE /{project}/api/result_cache/{hash}``. +""" + +from __future__ import annotations + +import shutil + +import pyarrow as pa +import pyarrow.parquet as pq +from fastapi.testclient import TestClient + +from tallyman_companion import create_app +from tallyman_core import entry_dir, read_manifest +from tallyman_core.errors import list_errors, record_error +from tallyman_core.paths import compute_cache_dir +from tallyman_xorq import build_and_persist +from tallyman_xorq.result_cache import baked_snapshot_path, cached_result_expr, snapshot_file_digest + + +def _agg_code(project: str) -> str: + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.group_by("region").aggregate(total=t.price.sum(), n=t.count()) +""" + + +def _second_agg_code(project: str) -> str: + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.group_by("category").aggregate(n=t.count()) +""" + + +def _nonreproducible_code(project: str) -> str: + """A recipe whose UDF returns other values on every call: a worthy entry (a UDF always is) that cannot be + reproduced, which ADR-009 D6 finds when the entry is created by running its query twice.""" + return f""" +from tallyman_xorq.io import read_project_file +from xorq.expr.udf import make_pandas_udf +import xorq.vendor.ibis.expr.datatypes as dt +from xorq.vendor.ibis import schema as ibis_schema + +t = read_project_file("orders.parquet", project={project!r}) + + +def jitter(df): + import random + + return df["qty"] + random.random() + + +_udf = make_pandas_udf(jitter, ibis_schema({{"qty": dt.int64}}), dt.float64, name="jitter") +expr = t.mutate(qty_jitter=_udf.on_expr(t)) +""" + + +def _cache_rows(client: TestClient, project: str) -> dict[str, dict]: + r = client.get(f"/{project}/api/result_cache") + assert r.status_code == 200, r.text + return {row["hash"]: row for row in r.json()["entries"]} + + +# --------------------------------------------------------------------------- +# nothing writes a file speculatively (ADR-007 D12, D5) +# --------------------------------------------------------------------------- + + +def test_startup_warm_up_writes_no_file(project, orders_parquet): + """ADR-007 D12: the warm-up used to call ``cached_result_expr`` for every entry until a 3 s budget was spent, + which heals a deleted snapshot. That undoes the Cache page's delete button, and one large heal blocks startup for + as long as it takes. With ``compute_cache/`` emptied and no request made, starting the app leaves it empty.""" + h = build_and_persist(project, _agg_code(project)).content_hash + snap = baked_snapshot_path(project, h) + assert snap is not None and snap.exists() + shutil.rmtree(compute_cache_dir(project), ignore_errors=True) + cached_result_expr.cache_clear() + + with TestClient(create_app(project)): # entering runs the startup handlers, including the warm-up + pass + + written = sorted(str(p) for p in compute_cache_dir(project).rglob("*") if p.is_file()) + assert written == [], "starting the app wrote files under compute_cache/" + assert not snap.exists() + + +def test_verify_sweep_reports_a_missing_snapshot_as_absent_and_writes_nothing(project, orders_parquet): + """ADR-007 D5 and D12: the verify sweep checks the files that exist and reports each entry that recorded a digest + as faithful, unfaithful or absent. It is not a caller of ``ensure_materialized``: a sweep that rewrote every + deleted snapshot in the project would undo the user's deletes. Every file that function writes is verified before + it is served, so an absent file is checked at the moment it next exists.""" + from tallyman_xorq.staleness import verify_sweep + + kept = build_and_persist(project, _agg_code(project)).content_hash + dropped = build_and_persist(project, _second_agg_code(project)).content_hash + snap = baked_snapshot_path(project, dropped) + assert snap is not None + snap.unlink() + cached_result_expr.cache_clear() + + out = verify_sweep(project) + + assert out["results"][kept] is True + assert "absent" in out, f"the sweep does not report absent files: {sorted(out)}" + assert sorted(out["absent"]) == [dropped] + assert out["unfaithful"] == [] and out["errors"] == {} + assert not snap.exists(), "the sweep rewrote a snapshot the user had deleted" + + +# --------------------------------------------------------------------------- +# the Cache page (ADR-007 D12, D14; ADR-009 D6) +# --------------------------------------------------------------------------- + + +def test_cache_page_lists_a_reproducible_entry_as_not_pinned(project, orders_parquet): + """ADR-009 D6: an entry whose query gave the same content digest twice is reproducible, so its file may be + deleted and made again. The Cache page says so for each row: ``pinned`` is False and there is no reason.""" + h = build_and_persist(project, _agg_code(project)).content_hash + client = TestClient(create_app(project)) + + row = _cache_rows(client, project)[h] + + assert row.get("pinned") is False, row + assert row.get("pinned_reason") is None, row + assert not row.get("orphan") + + +def test_a_non_reproducible_entry_is_pinned_and_its_delete_is_refused(project, orders_parquet): + """ADR-009 D6 and ADR-007 D12: an entry whose recipe is not reproducible is recorded as such when it is created, + and its file is never deleted by tallyman, because deleting it would end the only copy of those rows. The Cache + page's delete answers 409, says why, and leaves the file.""" + result = build_and_persist(project, _nonreproducible_code(project)) + h = result.content_hash + snap = baked_snapshot_path(project, h) + assert snap is not None and snap.exists() + client = TestClient(create_app(project)) + + row = _cache_rows(client, project)[h] + assert row.get("pinned") is True, row + assert row.get("pinned_reason"), row + + response = client.delete(f"/{project}/api/result_cache/{h}") + + assert response.status_code == 409, response.text + assert "reproducible" in response.json()["detail"].lower() + assert snap.exists(), "the delete removed the file of an entry that cannot be recreated" + + +def test_an_entry_with_an_unfaithful_heal_record_is_pinned(project, orders_parquet): + """ADR-007 D12 and ADR-006 D12 (unfaithful entries are pinned and badged): a heal that wrote different rows than + were built leaves an ``unfaithful_heal`` record in ``errors.jsonl``. That record is what pins the entry, so the + Cache page's delete refuses it as it refuses an entry that was found not reproducible at creation.""" + h = build_and_persist(project, _agg_code(project)).content_hash + snap = baked_snapshot_path(project, h) + assert snap is not None and snap.exists() + record_error(project, code="unfaithful_heal", message="self-heal produced different bytes", hash=h) + client = TestClient(create_app(project)) + + row = _cache_rows(client, project)[h] + assert row.get("pinned") is True, row + assert row.get("pinned_reason"), row + + assert client.delete(f"/{project}/api/result_cache/{h}").status_code == 409 + assert snap.exists() + + +def test_a_snapshot_whose_entry_is_not_in_the_catalog_is_listed_and_can_be_deleted(project, orders_parquet): + """ADR-007 D14: a reset no longer prunes ``compute_cache/``, so the snapshot of an entry that a reset retired stays + on disk until the user deletes it. The Cache page lists files by entry, so it needs a row for a file whose entry is + not in the catalog, and the delete has to accept it.""" + h = build_and_persist(project, _agg_code(project)).content_hash + snapshots = compute_cache_dir(project) / "result_cache" + snapshots.mkdir(parents=True, exist_ok=True) + orphan = snapshots / "deadbeef0123.parquet" + pq.write_table(pa.table({"a": [1, 2, 3]}), orphan) + client = TestClient(create_app(project)) + + rows = _cache_rows(client, project) + + assert "deadbeef0123" in rows, f"the file of an entry that is not in the catalog is not listed: {sorted(rows)}" + assert rows["deadbeef0123"].get("orphan") is True + assert rows["deadbeef0123"].get("alias") is None + assert not rows[h].get("orphan") + + response = client.delete(f"/{project}/api/result_cache/deadbeef0123") + + assert response.status_code == 200, response.text + assert not orphan.exists() + + +def test_a_normal_snapshot_can_still_be_deleted_and_the_next_read_re_creates_and_verifies_it(project, orders_parquet): + """ADR-007 D12 and D5: the delete button keeps working for an ordinary entry. The next read rewrites the file and + checks it against the recorded digest (no ``unfaithful_heal`` record), and the entry is still not pinned.""" + h = build_and_persist(project, _agg_code(project)).content_hash + snap = baked_snapshot_path(project, h) + assert snap is not None and snap.exists() + client = TestClient(create_app(project)) + + response = client.delete(f"/{project}/api/result_cache/{h}") + assert response.status_code == 200, response.text + assert not snap.exists() + + cached_result_expr.cache_clear() + assert len(cached_result_expr(project, h).execute()) > 0 + assert snap.exists() + assert snapshot_file_digest(snap) == read_manifest(entry_dir(project, h)).result_digest + assert not [e for e in list_errors(project) if e.get("code") == "unfaithful_heal"] + assert _cache_rows(client, project)[h].get("pinned") is False diff --git a/tests/test_digest.py b/tests/test_digest.py new file mode 100644 index 0000000..cd24900 --- /dev/null +++ b/tests/test_digest.py @@ -0,0 +1,281 @@ +"""ADR-009 D2: ``result_digest`` is a content digest of the snapshot, computed from the file read back. + +``snapshot_file_digest(path)`` today is the SHA-256 of the file's bytes, so it moves with the writer's row-group size, +codec, encoding and version, none of which are the result. ADR-009 decision D2 (result_digest is a digest of the +snapshot's content) replaces it with an order-sensitive digest over the Arrow data read back from the file, stored as +``arrow-sha256:``. These tests pin that definition on parquet files written directly with pyarrow, so no entry +needs to be built. + +Every test goes through ``_digest``, which also asserts the ``arrow-sha256:`` prefix. That makes each test a red test +today for the reason the ADR gives (the digest is not a content digest), including the "sees what it must" tests, +which would otherwise pass on a byte hash for the wrong reason. + +The per-column digests named by ``tallyman_xorq.digest.column_digests`` (the module is new) are what ADR-009 decision +D6 (create runs the query twice and compares) uses to name the columns whose values differ between two runs. +""" + +from __future__ import annotations + +from pathlib import Path + +import numpy as np +import pyarrow as pa +import pyarrow.parquet as pq +import pytest + +from tallyman_xorq.result_cache import snapshot_file_digest + +PREFIX = "arrow-sha256:" +N = 20_000 + +# The same rows written four ways. None of these settings is the result, so none may move the digest. +FORMATS = { + "snappy, 2,000-row groups": {"compression": "snappy", "row_group_size": 2_000}, + "zstd, one large group": { + "compression": "zstd", + "compression_level": 3, + "row_group_size": 1_048_576, + "version": "2.6", + "data_page_version": "1.0", + }, + "uncompressed, no dictionary, page v2, 7,500-row groups": { + "compression": "none", + "use_dictionary": False, + "data_page_version": "2.0", + "row_group_size": 7_500, + }, + "snappy, 999-row groups": {"compression": "snappy", "row_group_size": 999}, +} + + +def _digest(path: Path) -> str: + """The file's digest, asserted to be an ``arrow-sha256:<64 hex>`` content digest (ADR-009 D2).""" + d = snapshot_file_digest(path) + assert d.startswith(PREFIX), f"expected an {PREFIX} content digest, got {d!r}" + hexpart = d[len(PREFIX) :] + assert len(hexpart) == 64 and all(c in "0123456789abcdef" for c in hexpart), d + return d + + +def _column_digests(path: Path) -> dict[str, str]: + import tallyman_xorq.digest as digest_module + + return digest_module.column_digests(path) + + +def _write(table: pa.Table, path: Path, **settings) -> Path: + pq.write_table(table, path, **settings) + return path + + +def _table() -> pa.Table: + """Ints, floats with NaNs and nulls, strings with nulls, booleans and timestamps: the shapes a snapshot holds.""" + rng = np.random.default_rng(11) + ints = rng.integers(0, 1_000_000, N) + floats = rng.random(N) + floats[rng.integers(0, N, 40)] = np.nan + return pa.table( + { + "id": pa.array(np.arange(N)), + "g": pa.array(ints), + "f": pa.array(floats, mask=rng.random(N) < 0.02), + "s": pa.array([f"k{v % 500:04d}" for v in ints], mask=rng.random(N) < 0.02), + "b": pa.array(ints % 2 == 0), + "ts": pa.array(ints.astype("datetime64[s]")), + } + ) + + +def _replace(table: pa.Table, name: str, values: pa.Array | pa.ChunkedArray) -> pa.Table: + return table.set_column(table.schema.get_field_index(name), name, values) + + +def _with_changed_int(table: pa.Table) -> pa.Table: + g = table["g"].to_numpy().copy() + g[12_345] += 1 + return _replace(table, "g", pa.array(g)) + + +def _with_null_replaced_by_zero(table: pa.Table) -> pa.Table: + f = table["f"].combine_chunks() + first_null = int(np.flatnonzero(f.is_null().to_numpy(zero_copy_only=False))[0]) + patched = pa.concat_arrays([f.slice(0, first_null), pa.array([0.0]), f.slice(first_null + 1)]) + return _replace(table, "f", patched) + + +def _with_first_two_rows_swapped(table: pa.Table) -> pa.Table: + return table.take(pa.array(np.r_[1, 0, np.arange(2, N)])) + + +# --------------------------------------------------------------------------- +# the definition: a content digest, prefixed with its algorithm +# --------------------------------------------------------------------------- + + +def test_digest_carries_its_algorithm_prefix(tmp_path): + """ADR-009 D2 (result_digest is a digest of the snapshot's content): stored as ``arrow-sha256:``. + + The prefix means a future definition can never be compared against this one by accident. + """ + assert _digest(_write(_table(), tmp_path / "a.parquet")) == _digest(_write(_table(), tmp_path / "b.parquet")) + + +def test_digest_ignores_row_group_size_codec_and_encoding(tmp_path): + """ADR-009 D2 (digest ignores batching and format): the same rows written four ways give one digest. + + Today the digest is a hash of the file's bytes, so each of these files gets its own. + """ + table = _table() + digests = { + label: _digest(_write(table, tmp_path / f"{i}.parquet", **kw)) for i, (label, kw) in enumerate(FORMATS.items()) + } + assert len(set(digests.values())) == 1, digests + + +def test_digest_treats_string_and_large_string_as_one_type(tmp_path): + """ADR-009 D2: each column stream is seeded with its logical type; string and large_string are one type.""" + table = _table() + wide_fields = [pa.field(f.name, pa.large_string() if pa.types.is_string(f.type) else f.type) for f in table.schema] + wide = table.cast(pa.schema(wide_fields)) + narrow_path = _write(table, tmp_path / "narrow.parquet") + wide_path = _write(wide, tmp_path / "wide.parquet") + assert pq.ParquetFile(narrow_path).schema_arrow.field("s").type == pa.string() + assert pq.ParquetFile(wide_path).schema_arrow.field("s").type == pa.large_string() + assert _digest(narrow_path) == _digest(wide_path) + + +def test_digest_ignores_how_nulls_are_encoded_in_the_file(tmp_path): + """ADR-009 D2: null slots are zeroed or emptied before hashing, because a null slot may hold anything. + + A parquet file cannot carry garbage in a null slot, but a reader may hand back different bytes there depending on + how the column was encoded (dictionary or plain) and paged (v1 or v2). Those must not reach the digest. + """ + table = _table() + plain = _write(table, tmp_path / "plain.parquet", use_dictionary=False, data_page_version="2.0", compression="none") + dictionary = _write( + table, tmp_path / "dict.parquet", use_dictionary=True, data_page_version="1.0", compression="zstd" + ) + assert table["f"].null_count > 0 and table["s"].null_count > 0 + assert _digest(plain) == _digest(dictionary) + + +def test_digest_of_an_empty_file_is_defined_and_depends_on_the_schema(tmp_path): + """ADR-009 D2: the streams are combined in schema order together with the row count, so zero rows still digest.""" + empty = _table().slice(0, 0) + a = _digest(_write(empty, tmp_path / "a.parquet", compression="snappy")) + b = _digest(_write(empty, tmp_path / "b.parquet", compression="zstd")) + one_row = _digest(_write(_table().slice(0, 1), tmp_path / "one.parquet")) + other_schema = _digest(_write(empty.select(["id", "g"]), tmp_path / "other.parquet")) + assert a == b + assert a != one_row + assert a != other_schema + + +# --------------------------------------------------------------------------- +# what the digest must see +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize( + "change", + [_with_changed_int, _with_null_replaced_by_zero, _with_first_two_rows_swapped], + ids=["one value changed", "one null replaced by 0.0", "first two rows swapped"], +) +def test_digest_sees_what_it_must(tmp_path, change): + """ADR-009 D2 (digest sees what it must): a changed value, a null turned into 0.0 and two swapped rows each move it. + + The digest stays order-sensitive, so the canonical sort is still what makes a snapshot reproducible. + """ + table = _table() + reference = _digest(_write(table, tmp_path / "reference.parquet")) + assert _digest(_write(change(table), tmp_path / "changed.parquet")) != reference + + +def test_digest_covers_column_names_and_types(tmp_path): + """ADR-009 D2: each stream is seeded with the column's name and logical type, so neither can change silently.""" + table = _table() + reference = _digest(_write(table, tmp_path / "reference.parquet")) + renamed = table.rename_columns(["id", "g2", "f", "s", "b", "ts"]) + narrowed = _replace(table, "g", table["g"].cast(pa.int32())) + assert _digest(_write(renamed, tmp_path / "renamed.parquet")) != reference + assert _digest(_write(narrowed, tmp_path / "narrowed.parquet")) != reference + + +# --------------------------------------------------------------------------- +# nested types: the spike covered fixed-width, boolean, string and binary only (ADR-009 open question 1) +# --------------------------------------------------------------------------- + + +def _nested_table(changed: str | None = None) -> pa.Table: + """A list, a struct and a map column, each with nulls; ``changed`` alters one value inside that column.""" + n = 4_000 + lists = [[i, i + 1 + (changed == "l" and i == 100)] if i % 7 else None for i in range(n)] + structs = [ + {"a": i, "b": f"x{i}" if (changed == "st" and i == 200) else f"s{i}"} if i % 11 else None for i in range(n) + ] + maps = [[("k", i), ("z", i * 2 + (changed == "m" and i == 300))] if i % 13 else None for i in range(n)] + return pa.table( + { + "l": pa.array(lists, pa.list_(pa.int64())), + "st": pa.array(structs, pa.struct([("a", pa.int64()), ("b", pa.string())])), + "m": pa.array(maps, pa.map_(pa.string(), pa.int64())), + } + ) + + +def test_nested_columns_digest_deterministically_whatever_the_format(tmp_path): + """ADR-009 D2: lists, structs and maps need a recursive definition; theirs must also ignore the file format.""" + table = _nested_table() + a = _digest(_write(table, tmp_path / "a.parquet", compression="snappy", row_group_size=500)) + b = _digest(_write(table, tmp_path / "b.parquet", compression="zstd", row_group_size=1_048_576)) + again = _digest(_write(_nested_table(), tmp_path / "c.parquet", compression="snappy", row_group_size=500)) + assert a == b == again + + +@pytest.mark.parametrize("column", ["l", "st", "m"]) +def test_nested_digest_sees_a_change_inside_the_nested_value(tmp_path, column): + """ADR-009 D2: one element changed inside a list, a struct field or a map value moves the digest.""" + reference = _digest(_write(_nested_table(), tmp_path / "reference.parquet")) + assert _digest(_write(_nested_table(changed=column), tmp_path / "changed.parquet")) != reference + + +# --------------------------------------------------------------------------- +# per-column digests: how a non-reproducible entry names its offending columns +# --------------------------------------------------------------------------- + + +def test_column_digests_are_keyed_by_column_in_schema_order(tmp_path): + """ADR-009 D6 (create runs the query twice and compares) needs a digest per column, keyed by name.""" + table = _table() + digests = _column_digests(_write(table, tmp_path / "a.parquet")) + assert list(digests) == table.schema.names + assert all(isinstance(v, str) and v for v in digests.values()) + + +@pytest.mark.parametrize( + ("change", "column"), + [(_with_changed_int, "g"), (_with_null_replaced_by_zero, "f")], + ids=["an int column", "a float column with a null replaced"], +) +def test_column_digests_name_exactly_the_changed_column(tmp_path, change, column): + """ADR-009 D6: two files that differ in one column differ, per column, in that column only.""" + table = _table() + reference = _column_digests(_write(table, tmp_path / "reference.parquet")) + changed = _column_digests(_write(change(table), tmp_path / "changed.parquet")) + assert {name for name in reference if reference[name] != changed[name]} == {column} + assert _digest(_write(table, tmp_path / "r2.parquet")) != _digest(_write(change(table), tmp_path / "c2.parquet")) + + +def test_column_digests_ignore_the_file_format(tmp_path): + """ADR-009 D2 and D6: the per-column digests are as format-independent as the whole-file digest.""" + table = _table() + a = _column_digests(_write(table, tmp_path / "a.parquet", compression="snappy", row_group_size=2_000)) + b = _column_digests(_write(table, tmp_path / "b.parquet", compression="zstd", row_group_size=1_048_576)) + assert a == b + + +def test_column_digests_of_nested_columns_name_the_changed_column(tmp_path): + """ADR-009 D6: a change inside a struct column is attributed to that column, not to its neighbours.""" + reference = _column_digests(_write(_nested_table(), tmp_path / "reference.parquet")) + changed = _column_digests(_write(_nested_table(changed="st"), tmp_path / "changed.parquet")) + assert {name for name in reference if reference[name] != changed[name]} == {"st"} diff --git a/tests/test_materialize.py b/tests/test_materialize.py new file mode 100644 index 0000000..03ca38e --- /dev/null +++ b/tests/test_materialize.py @@ -0,0 +1,546 @@ +"""ADR-007: tallyman owns result materialization (no xorq cache nodes in builds). + +Covers D1 (builds carry no cache nodes), D2 (a snapshot's location is a function of the content hash), D3 (chaining +through a worthy parent is a bare read of its snapshot), D4 (one writer, used by the build and by every heal), D5 +(``ensure_materialized`` makes files exist before anything runs), D7 (the cold state is an empty ``compute_cache``), +D8 (a sentinel keeps xorq's cache directory empty) and D13 (a file is cache only if ``ensure_materialized`` can +re-create it). ``plans/ADR-007-tallyman-owned-materialization.md`` has the decisions; the shared API is fixed by the +lead's contract, so the names imported lazily below do not exist yet and these tests are red until they do. +""" + +from __future__ import annotations + +import json +import os +import shutil +from pathlib import Path + +import httpx +import pyarrow as pa +import pyarrow.parquet as pq +import pytest + +from tallyman_cli.fixtures import write_shoe_orders +from tallyman_companion.diff import build_diff_expr +from tallyman_core import data_dir, entry_dir +from tallyman_core.manifest import read_manifest +from tallyman_core.paths import compute_cache_dir, tallyman_home +from tallyman_mcp.server import catalog_create, catalog_revise +from tallyman_xorq import result_cache +from tallyman_xorq.build import BuildError, build_and_persist +from tallyman_xorq.result_cache import baked_snapshot_path, cached_result_expr, snapshot_file_digest + +# --------------------------------------------------------------------------- # +# recipes +# --------------------------------------------------------------------------- # + + +def _agg_code(project: str) -> str: # an Aggregate: worthy, so it has a snapshot + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.group_by("region").aggregate(total=t.price.sum(), n=t.count()) +""" + + +def _boots_agg_code(project: str) -> str: # a second version of the same aggregate, over fewer rows + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +b = t.filter(t.category == "boots") +expr = b.group_by("region").aggregate(total=b.price.sum(), n=b.count()) +""" + + +def _cheap_child_code(parent: str) -> str: # a filter and a computed column over a worthy parent's result + return f""" +from tallyman_xorq.io import tracked_expr_from_alias +t = tracked_expr_from_alias({parent!r}) +expr = t.filter(t.n > 0).mutate(share=t.total / t.n) +""" + + +def _worthy_grandchild_code(parent: str) -> str: # an Aggregate over the cheap child + return f""" +from tallyman_xorq.io import tracked_expr_from_alias +t = tracked_expr_from_alias({parent!r}) +expr = t.aggregate(total_share=t.share.sum(), rows=t.count()) +""" + + +def _root_code(project: str) -> str: # a bare read of a source: a cheap root entry + return f""" +from tallyman_xorq.io import read_project_file +expr = read_project_file("orders.parquet", project={project!r}) +""" + + +def _root_child_code(alias: str) -> str: # a cheap child over a cheap root, so it inlines the root's graph + return f""" +from tallyman_xorq.io import tracked_expr_from_alias +t = tracked_expr_from_alias({alias!r}) +expr = t.filter(t.qty > 1) +""" + + +def _hash(res: dict) -> str: + assert "error" not in res, res + return res["hash"] + + +def _digest_of(project: str, content_hash: str) -> str | None: + return read_manifest(entry_dir(project, content_hash)).result_digest + + +def _build_yaml(project: str, content_hash: str) -> str: + return (entry_dir(project, content_hash) / "xorq_build" / "expr.yaml").read_text() + + +@pytest.fixture(autouse=True) +def _no_cascade(monkeypatch): + """These tests revise aliases to get a second version; keep the cascade out of them.""" + monkeypatch.setenv("TALLYMAN_AUTO_RECALC", "0") + + +# --------------------------------------------------------------------------- # +# D1: builds carry no cache nodes +# --------------------------------------------------------------------------- # + + +def test_a_recipe_that_calls_cache_is_a_build_error(project, orders_parquet): + """ADR-007 D1 (builds carry no cache nodes): a ``CachedNode`` in a recipe would write under ~/.cache/xorq.""" + code = f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.select("region", "price").cache() +""" + with pytest.raises(BuildError, match="cache"): + build_and_persist(project, code) + + +def test_a_build_carries_no_cache_node(project, orders_parquet, monkeypatch): + """ADR-007 D1: neither a worthy entry nor its chained child has a ``CachedNode`` in its frozen build.""" + monkeypatch.setenv("TALLYMAN_PROJECT", project) + parent = _hash(catalog_create("agg", _agg_code(project))) + child = _hash(catalog_create("child", _cheap_child_code("agg"))) + for h in (parent, child): + assert "CachedNode" not in _build_yaml(project, h) + + +def test_the_xorq_cache_machinery_is_gone(): + """ADR-007 Consequences (retired): the code whose only job was to aim xorq's cache is deleted.""" + import tallyman_xorq.portable as portable + import tallyman_xorq.source_cache as source_cache + from tallyman_core.manifest import Manifest + + assert not hasattr(portable, "rewrite_cache_dirs") + for name in ("_cached_node_path", "_assert_recorded_snapshot_key", "entry_graph_expr", "classify_build"): + assert not hasattr(result_cache, name), name + assert not hasattr(source_cache, "_is_worthy_expr") + assert "snapshot_key" not in Manifest.model_fields + + +# --------------------------------------------------------------------------- # +# D2: a snapshot's location is a function of the content hash +# --------------------------------------------------------------------------- # + + +def test_the_snapshot_path_is_derived_from_the_content_hash(project, orders_parquet, monkeypatch): + """ADR-007 D2 (snapshot location): ``compute_cache/result_cache/.parquet``, written at create.""" + from tallyman_xorq.materialize import snapshot_path + + monkeypatch.setenv("TALLYMAN_PROJECT", project) + h = _hash(catalog_create("agg", _agg_code(project))) + expected = compute_cache_dir(project) / "result_cache" / f"{h}.parquet" + assert snapshot_path(project, h) == expected + assert expected.is_file() + assert baked_snapshot_path(project, h) == expected + + +def test_the_manifest_records_no_snapshot_key(project, orders_parquet, monkeypatch): + """ADR-007 D2: with the path computed from the hash there is no second derivation for a tripwire to compare.""" + monkeypatch.setenv("TALLYMAN_PROJECT", project) + h = _hash(catalog_create("agg", _agg_code(project))) + assert "snapshot_key" not in json.loads((entry_dir(project, h) / "manifest.json").read_text()) + + +def test_a_worthy_read_is_one_memoised_bare_read_of_the_snapshot(project, orders_parquet, monkeypatch): + """ADR-007 D2: ``cached_result_expr`` is one bare read of the snapshot, memoised, and loads no build.""" + from tallyman_xorq.materialize import snapshot_path + + monkeypatch.setenv("TALLYMAN_PROJECT", project) + h = _hash(catalog_create("agg", _agg_code(project))) + cached_result_expr.cache_clear() + + def _no_build_load(*a, **k): + raise AssertionError("a worthy entry whose snapshot exists must be served without loading its build") + + monkeypatch.setattr(result_cache, "load_entry_expr", _no_build_load) + first = cached_result_expr(project, h) + assert type(first.op()).__name__ == "Read" + assert str(snapshot_path(project, h)) in str(dict(first.op().read_kwargs).get("hash_path")) + assert cached_result_expr(project, h) is first # one read has one table name + + +# --------------------------------------------------------------------------- # +# D3: chaining through a worthy parent is a bare read of its snapshot +# --------------------------------------------------------------------------- # + + +def test_a_child_of_a_worthy_parent_reads_the_parents_snapshot_and_is_cheap(project, orders_parquet, monkeypatch): + """ADR-007 D3 (chaining is a bare read): a filter over an aggregate's snapshot is cheap, not a second copy.""" + monkeypatch.setenv("TALLYMAN_PROJECT", project) + parent = _hash(catalog_create("agg", _agg_code(project))) + child = _hash(catalog_create("child", _cheap_child_code("agg"))) + yaml_text = _build_yaml(project, child) + assert f"result_cache/{parent}.parquet" in yaml_text + assert "op: Aggregate" not in yaml_text + manifest = read_manifest(entry_dir(project, child)) + assert manifest.cache_worthy is False + assert baked_snapshot_path(project, child) is None + + +def test_a_child_hash_follows_the_parents_snapshot_path_not_its_bytes(project, orders_parquet, monkeypatch): + """ADR-007 D3: content identity goes in the path. Other bytes at the same path keep the child's hash; a new + parent (a new path) changes it.""" + from tallyman_xorq.materialize import snapshot_path + + monkeypatch.setenv("TALLYMAN_PROJECT", project) + parent = _hash(catalog_create("agg", _agg_code(project))) + child = _hash(catalog_create("child", _cheap_child_code("agg"))) + + snap = snapshot_path(project, parent) + table = pq.read_table(snap) + tampered = table.set_column(table.schema.get_field_index("total"), "total", pa.array([1.0] * table.num_rows)) + pq.write_table(tampered, snap) + cached_result_expr.cache_clear() + assert build_and_persist(project, _cheap_child_code("agg")).content_hash == child + + catalog_revise("agg", _boots_agg_code(project)) # a new parent entry, so a new snapshot path + assert build_and_persist(project, _cheap_child_code("agg")).content_hash != child + + +# --------------------------------------------------------------------------- # +# D4: one writer +# --------------------------------------------------------------------------- # + + +def test_a_create_replaces_a_snapshot_that_is_already_on_disk(project, orders_parquet): + """ADR-007 D4 (one writer) and D14: a create never looks for the file, it runs the query and replaces it.""" + from tallyman_xorq.materialize import snapshot_path + + code = _agg_code(project) + first = build_and_persist(project, code) + snap = snapshot_path(project, first.content_hash) + good = snapshot_file_digest(snap) + shutil.rmtree(entry_dir(project, first.content_hash)) # the entry goes (a reset), its file stays + snap.write_bytes(b"this is not parquet") + + again = build_and_persist(project, code) + assert again.content_hash == first.content_hash + assert snapshot_file_digest(snap) == good + + +def test_materialize_leaves_one_file_and_no_temp_files(project, orders_parquet): + """ADR-007 D4: a unique temp name in the destination directory, then ``os.replace``.""" + from tallyman_xorq.materialize import materialize, snapshot_path + + h = build_and_persist(project, _agg_code(project)).content_hash + result = materialize(project, h) + snap = snapshot_path(project, h) + assert result.path == snap + assert result.digest == _digest_of(project, h) + assert sorted(p.name for p in snap.parent.iterdir()) == [snap.name] + + +def test_a_failed_materialization_keeps_the_previous_file(project, orders_parquet, monkeypatch): + """ADR-007 D4: the destination only ever changes by an atomic replace of a complete file.""" + from tallyman_xorq.materialize import materialize, snapshot_path + + h = build_and_persist(project, _agg_code(project)).content_hash + snap = snapshot_path(project, h) + before = snap.read_bytes() + + def _boom(self, *args, **kwargs): + raise OSError("disk full") + + monkeypatch.setattr(pq.ParquetWriter, "write_table", _boom) + with pytest.raises(OSError, match="disk full"): + materialize(project, h) + assert snap.read_bytes() == before + assert sorted(p.name for p in snap.parent.iterdir()) == [snap.name] + + +# --------------------------------------------------------------------------- # +# D5: ensure_materialized +# --------------------------------------------------------------------------- # + + +def test_ensure_materialized_rewrites_a_missing_snapshot_and_verifies_it(project, orders_parquet): + """ADR-007 D5 (``ensure_materialized``): a deleted snapshot is re-created and checked against the manifest.""" + from tallyman_xorq.materialize import ensure_materialized, snapshot_path + + h = build_and_persist(project, _agg_code(project)).content_hash + snap = snapshot_path(project, h) + snap.unlink() + ensure_materialized(project, h) + assert snap.is_file() + recorded = _digest_of(project, h) + assert recorded and recorded.startswith("arrow-sha256:") + assert snapshot_file_digest(snap) == recorded + + +def test_files_exist_before_anything_runs(project, orders_parquet, monkeypatch): + """ADR-007 D5: with an ancestor's snapshot deleted, opening a descendant rewrites the ancestor first, so no plan + is ever executed over a file that is missing. This is #76's reproduction.""" + from tallyman_xorq.materialize import snapshot_path + + monkeypatch.setenv("TALLYMAN_PROJECT", project) + parent = _hash(catalog_create("agg", _agg_code(project))) + child = _hash(catalog_create("child", _cheap_child_code("agg"))) + snapshot_path(project, parent).unlink() + cached_result_expr.cache_clear() + + expr = cached_result_expr(project, child) # composing, not executing + assert snapshot_path(project, parent).is_file() + assert snapshot_file_digest(snapshot_path(project, parent)) == _digest_of(project, parent) + assert len(expr.execute()) > 0 + + +def test_building_a_child_rewrites_a_deleted_parent_snapshot_first(project, orders_parquet, monkeypatch): + """ADR-007 D3 and D5: chaining at mint time makes the parent's file exist, since ``build_expr`` of a child fails + while the parent's snapshot is absent.""" + from tallyman_xorq.materialize import snapshot_path + + monkeypatch.setenv("TALLYMAN_PROJECT", project) + parent = _hash(catalog_create("agg", _agg_code(project))) + snapshot_path(project, parent).unlink() + cached_result_expr.cache_clear() + + child = build_and_persist(project, _cheap_child_code("agg")).content_hash + assert snapshot_path(project, parent).is_file() + assert f"result_cache/{parent}.parquet" in _build_yaml(project, child) + + +def test_composing_a_diff_rewrites_both_deleted_snapshots_first(project, orders_parquet, monkeypatch): + """ADR-007 D5 and D10 (diffs: what this set still does): both sides are read through ``cached_result_expr``, so + both files exist before the join is composed, and the diff carries no row-order column from either side.""" + from tallyman_xorq.materialize import snapshot_path + + monkeypatch.setenv("TALLYMAN_PROJECT", project) + a = _hash(catalog_create("agg", _agg_code(project))) + b = _hash(catalog_revise("agg", _boots_agg_code(project))) + for h in (a, b): + snapshot_path(project, h).unlink() + cached_result_expr.cache_clear() + + diff = build_diff_expr(a, b, keys=["region"]) + for h in (a, b): + assert snapshot_path(project, h).is_file() + assert snapshot_file_digest(snapshot_path(project, h)) == _digest_of(project, h) + # The inputs carry the column, so the absence below is a decision and not an accident. + assert cached_result_expr(project, a).columns[-1] == "__row_order" + assert cached_result_expr(project, b).columns[-1] == "__row_order" + assert not [c for c in diff.columns if c.startswith("__row_order")], list(diff.columns) + assert len(diff.execute()) > 0 + + +# --------------------------------------------------------------------------- # +# D7: the cold state is an empty compute_cache +# --------------------------------------------------------------------------- # + + +def test_an_empty_compute_cache_reproduces_every_snapshot_an_entry_needs(project, orders_parquet, monkeypatch): + """ADR-007 D7 (the cold seam): with ``compute_cache/`` removed the canonical read reproduces every snapshot the + entry needs, each with its recorded digest, ordered copies of sources included.""" + from tallyman_xorq.materialize import snapshot_path + + monkeypatch.setenv("TALLYMAN_PROJECT", project) + parent = _hash(catalog_create("agg", _agg_code(project))) + child = _hash(catalog_create("child", _cheap_child_code("agg"))) + grandchild = _hash(catalog_create("grand", _worthy_grandchild_code("child"))) + assert child != parent + recorded = {h: _digest_of(project, h) for h in (parent, grandchild)} + assert all(d and d.startswith("arrow-sha256:") for d in recorded.values()) + + shutil.rmtree(compute_cache_dir(project)) + cached_result_expr.cache_clear() + assert len(cached_result_expr(project, grandchild).execute()) == 1 + + for h, digest in recorded.items(): + assert snapshot_file_digest(snapshot_path(project, h)) == digest, h + + +# --------------------------------------------------------------------------- # +# D8: a sentinel keeps xorq's cache directory empty +# --------------------------------------------------------------------------- # + + +class _AliveProc: + def poll(self): + return None + + +def _fake_buckaroo(): + """A ``BuckarooManager`` whose subprocess is 'alive' and whose HTTP client accepts every ``/load_expr``.""" + from tallyman_companion.buckaroo_lifecycle import BuckarooManager + + def handler(request: httpx.Request) -> httpx.Response: + body = json.loads(request.content or b"{}") + return httpx.Response(200, json={"session": body.get("session") or "s"}) + + bk = BuckarooManager() + bk.proc = _AliveProc() + bk.bound_port = 8799 + bk._maybe_restart = lambda: None + bk._client = httpx.Client(transport=httpx.MockTransport(handler)) + return bk + + +def _xorq_cache_files() -> set[str]: + # tests/conftest.py points the whole session at one XORQ_CACHE_DIR (xorq freezes it at first import), so the + # test compares the directory before and after instead of asserting it is empty. + root = Path(os.environ["XORQ_CACHE_DIR"]) + return {str(p) for p in root.rglob("*") if p.is_file()} + + +def test_xorq_cache_directory_stays_untouched(project, orders_parquet, monkeypatch): + """ADR-007 D8 (the sentinel): a build, a chained child build, a view, a delete and a reopen write nothing under + xorq's cache directory. Fails on main: a child build re-executes its parent into ``~/.cache/xorq``.""" + monkeypatch.setenv("TALLYMAN_PROJECT", project) + before = _xorq_cache_files() + + parent = _hash(catalog_create("agg", _agg_code(project))) + child = _hash(catalog_create("child", _cheap_child_code("agg"))) + assert len(cached_result_expr(project, child).execute()) > 0 + _fake_buckaroo().load_session(parent, project) # the view of a worthy entry + baked_snapshot_path(project, parent).unlink() # the user deletes the file... + cached_result_expr.cache_clear() + assert len(cached_result_expr(project, parent).execute()) > 0 # ...and reopens the entry + _fake_buckaroo().load_session(child, project) + + assert _xorq_cache_files() == before + + +# --------------------------------------------------------------------------- # +# D13: a file is cache only if ensure_materialized can re-create it +# --------------------------------------------------------------------------- # + + +def _ordered_copies(project: str) -> list[Path]: + return sorted((compute_cache_dir(project) / "ordered_sources").glob("*.parquet")) + + +def _clones(project: str) -> list[Path]: + cas = data_dir(project) / ".cas" + return sorted(cas.iterdir()) if cas.is_dir() else [] + + +def _rows(project: str, content_hash: str) -> list[dict]: + df = cached_result_expr(project, content_hash).execute().sort_values("__row_order") + return json.loads(df.to_json(orient="records")) + + +def test_the_ordered_copy_lives_under_the_compute_cache_and_is_recorded(project, orders_parquet, monkeypatch): + """ADR-007 D13 (ordered copies are cache): they live under ``compute_cache/``, ``csv_ordered/`` is retired, and + the manifest records what ``ensure_materialized`` needs to make one again.""" + monkeypatch.setenv("TALLYMAN_PROJECT", project) + h = _hash(catalog_create("orders", _root_code(project))) + copies = _ordered_copies(project) + assert len(copies) == 1 + assert not (tallyman_home() / "csv_ordered").exists() + record = read_manifest(entry_dir(project, h)).ordered_copies[copies[0].stem] + assert record["source"] == "orders.parquet" + assert record["reader"]["kind"] == "parquet" + assert record["digest"] == read_manifest(entry_dir(project, h)).sources["orders.parquet"] + assert record["content_digest"] == snapshot_file_digest(copies[0]) + + +def test_a_deleted_ordered_copy_is_made_again_from_the_clone_and_checked(project, orders_parquet, monkeypatch): + """ADR-007 D13 and D5: opening a root entry re-creates its ordered copy from the clone.""" + monkeypatch.setenv("TALLYMAN_PROJECT", project) + h = _hash(catalog_create("orders", _root_code(project))) + [copy] = _ordered_copies(project) + digest = snapshot_file_digest(copy) + rows = _rows(project, h) + copy.unlink() + cached_result_expr.cache_clear() + + assert _rows(project, h) == rows + assert snapshot_file_digest(copy) == digest + + +def test_a_deleted_clone_is_made_again_from_the_unchanged_live_source(project, orders_parquet, monkeypatch): + """ADR-007 D13: a clone can be copied again from the live source while the live bytes still hash to its name.""" + monkeypatch.setenv("TALLYMAN_PROJECT", project) + h = _hash(catalog_create("orders", _root_code(project))) + [copy] = _ordered_copies(project) + [clone] = _clones(project) + rows = _rows(project, h) + copy.unlink() + clone.unlink() + cached_result_expr.cache_clear() + + assert _rows(project, h) == rows + assert clone.is_file() and copy.is_file() + + +def test_a_child_that_inlines_a_cheap_parent_can_make_the_ordered_copy_again(project, orders_parquet, monkeypatch): + """ADR-007 D5 and D13: a child's build reads the parent's ordered copy directly, so the child's manifest carries + the record that lets it be made again.""" + monkeypatch.setenv("TALLYMAN_PROJECT", project) + _hash(catalog_create("orders", _root_code(project))) + child = _hash(catalog_create("multi", _root_child_code("orders"))) + [copy] = _ordered_copies(project) + [clone] = _clones(project) + rows = _rows(project, child) + copy.unlink() + clone.unlink() + cached_result_expr.cache_clear() + + assert _rows(project, child) == rows + assert copy.is_file() + + +def test_when_nothing_can_make_a_file_again_the_error_names_the_source(project, orders_parquet, monkeypatch): + """ADR-007 D13: the clone is gone and the live source has changed, so the built rows are unrecoverable and the + error names the source file.""" + monkeypatch.setenv("TALLYMAN_PROJECT", project) + h = _hash(catalog_create("orders", _root_code(project))) + [copy] = _ordered_copies(project) + [clone] = _clones(project) + copy.unlink() + clone.unlink() + write_shoe_orders(data_dir(project) / "orders.parquet", n_rows=37, seed=99) # the live source changed + cached_result_expr.cache_clear() + + with pytest.raises(Exception, match=r"orders\.parquet"): + cached_result_expr(project, h) + + +def test_a_csv_ordered_copy_is_made_again_with_the_recorded_reader_options(project, monkeypatch): + """ADR-007 D13: the manifest records each source's reader options, so a CSV copy is re-created with the schema and + the ``scan_csv`` options it was first read with.""" + monkeypatch.setenv("TALLYMAN_PROJECT", project) + csv = data_dir(project) / "sample.csv" + csv.write_text("id;name;value\n3;charlie;30\n1;alice;10\n2;bob;20\n") + code = f""" +import xorq.vendor.ibis as ibis +from tallyman_xorq.io import tallyman_read_csv +schema = ibis.schema({{"id": "int64", "name": "string", "value": "int64"}}) +expr = tallyman_read_csv({str(csv)!r}, schema=schema, separator=";") +""" + h = build_and_persist(project, code).content_hash + [copy] = _ordered_copies(project) + record = read_manifest(entry_dir(project, h)).ordered_copies[copy.stem] + assert record["reader"]["kind"] == "csv" + assert record["reader"]["scan_kwargs"] == {"separator": ";"} + rows = _rows(project, h) + assert [r["id"] for r in rows] == [3, 1, 2] # file order, not sorted + + copy.unlink() + [clone] = _clones(project) + clone.unlink() + cached_result_expr.cache_clear() + assert _rows(project, h) == rows + assert snapshot_file_digest(copy) == record["content_digest"] diff --git a/tests/test_project_lock.py b/tests/test_project_lock.py new file mode 100644 index 0000000..376373e --- /dev/null +++ b/tests/test_project_lock.py @@ -0,0 +1,262 @@ +"""One write at a time per project. + +Red tests for ``plans/ADR-007-tallyman-owned-materialization.md`` D11. Every write takes the project's existing lock +(``.checkpoint.lock``): a build, a materialization, a promote, a recalc and the checkpoint that takes it today. FastMCP +runs tool calls on a thread pool, so two parallel tool calls used to build at once, and "two builds of one entry can +end with the failing one deleting the winner's directory" (``build.py:447-468`` and ``618-627``). The lock has to be +re-entrant per thread, because a promote builds and then checkpoints, and it has to be a real lock between threads and +between processes. + +Threads here are always daemon threads joined with a timeout, so a regression shows up as a failed assertion and not as +a CI job that hangs. +""" + +from __future__ import annotations + +import threading + +from tallyman_core import entry_dir, read_manifest +from tallyman_xorq import build_and_persist + + +def _agg_code(project: str) -> str: + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.group_by("region").aggregate(total=t.price.sum(), n=t.count()) +""" + + +def _cheap_code(project: str) -> str: + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.filter(t.price > 0) +""" + + +def _run_in_threads(targets, join_timeout: float = 120.0) -> list[threading.Thread]: + threads = [threading.Thread(target=target, daemon=True) for target in targets] + for thread in threads: + thread.start() + for thread in threads: + thread.join(timeout=join_timeout) + return threads + + +# --------------------------------------------------------------------------- +# the lock itself +# --------------------------------------------------------------------------- + + +def test_nested_acquire_in_one_thread_does_not_block(project): + """ADR-007 D11 (one write at a time per project): a promote builds and then checkpoints, and a build + materializes, so a thread that holds the lock takes it again. ``_project_lock`` takes ``flock`` on a fresh file + descriptor, so today the nested acquire waits for the outer one, which is the same thread, forever.""" + from tallyman_core import catalog_state + + entered = threading.Event() + + def nested() -> None: + with catalog_state._project_lock(project): + with catalog_state._project_lock(project): + entered.set() + + (worker,) = _run_in_threads([nested], join_timeout=3) + + assert not worker.is_alive(), "a nested acquire of the project lock in one thread blocked" + assert entered.is_set() + + +def test_a_second_thread_waits_for_the_holder_and_proceeds_after_release(project): + """ADR-007 D11: making the lock re-entrant must stay per thread. A thread that does not hold it waits, and gets it + once the holder leaves. (This one already holds on today's code; it guards the re-entrant change.)""" + from tallyman_core import catalog_state + + holding, release, second_in = threading.Event(), threading.Event(), threading.Event() + + def holder() -> None: + with catalog_state._project_lock(project): + holding.set() + release.wait(timeout=10) + + def second() -> None: + holding.wait(timeout=10) + with catalog_state._project_lock(project): + second_in.set() + + threads = [threading.Thread(target=t, daemon=True) for t in (holder, second)] + for thread in threads: + thread.start() + assert holding.wait(timeout=5) + assert not second_in.wait(timeout=0.5), "a second thread entered while the first still held the project lock" + + release.set() + + assert second_in.wait(timeout=5), "the second thread never got the lock after the holder released it" + for thread in threads: + thread.join(timeout=5) + + +def test_project_lock_is_public_and_the_private_name_is_an_alias(): + """ADR-007 D11: ``build`` and ``materialize`` live outside ``catalog_state``, so the lock gets a public name. + ``_project_lock`` stays for the callers that exist.""" + from tallyman_core import catalog_state + + assert catalog_state.project_lock is catalog_state._project_lock + + +# --------------------------------------------------------------------------- +# what takes it +# --------------------------------------------------------------------------- + + +def test_two_builds_in_one_project_never_run_at_once(project, orders_parquet, monkeypatch): + """ADR-007 D11: a build is a write, so two builds of two different entries in one project queue. ``build_expr`` is + the first heavy step of a build; it is wrapped to count how many threads are inside it. Each waits (on an event, + not a sleep) for the other to be inside too. Unserialized, they meet at once and the peak is 2. Serialized, the + second cannot enter while the first is in, the wait times out, and the peak is 1.""" + import xorq.ibis_yaml.compiler as compiler + + real_build_expr = compiler.build_expr + guard = threading.Lock() + state = {"active": 0, "peak": 0} + both_inside = threading.Event() + + def instrumented(*args, **kwargs): + with guard: + state["active"] += 1 + state["peak"] = max(state["peak"], state["active"]) + if state["active"] >= 2: + both_inside.set() + try: + both_inside.wait(timeout=1.0) + return real_build_expr(*args, **kwargs) + finally: + with guard: + state["active"] -= 1 + + monkeypatch.setattr(compiler, "build_expr", instrumented) + start = threading.Barrier(2) + results: list = [] + errors: list[BaseException] = [] + + def builder(code: str): + def run() -> None: + try: + start.wait(timeout=10) + results.append(build_and_persist(project, code)) + except BaseException as exc: # noqa: BLE001 - reported by the assertion below + errors.append(exc) + + return run + + threads = _run_in_threads([builder(_agg_code(project)), builder(_cheap_code(project))]) + + assert not any(t.is_alive() for t in threads), "a build never finished" + assert state["peak"] == 1, f"{state['peak']} builds of one project ran at the same time" + assert errors == [] + assert len(results) == 2 and results[0].content_hash != results[1].content_hash + + +def test_a_failing_concurrent_build_of_one_entry_cannot_delete_the_winners_entry(project, orders_parquet, monkeypatch): + """ADR-007 D11 (the audit's finding): two builds of one entry, and one of them fails after laying down the entry + directory. Its cleanup removes the directory, which the other build is still using, so the winner then fails too. + + The failure is injected, so that this is not a race that may or may not happen: whichever thread reaches + ``load_expr`` first is the primary and waits (on an event, with a timeout) for a second thread to arrive, and a + second thread that does arrive raises, as a writer that lost a race for xorq's shared temp file does. Serialized, + the second thread never arrives (it waits for the lock, then finds a complete entry and returns), so nothing is + injected and both builds return the one entry.""" + import xorq.ibis_yaml.compiler as compiler + + real_load_expr = compiler.load_expr + guard = threading.Lock() + state: dict = {"primary": None} + second_arrived = threading.Event() + + def instrumented(*args, **kwargs): + me = threading.get_ident() + with guard: + first_call = state["primary"] is None + if first_call: + state["primary"] = me + is_primary = state["primary"] == me + if not is_primary: + second_arrived.set() + raise RuntimeError("simulated failure of a concurrent writer of the same entry") + if first_call: + second_arrived.wait(timeout=1.5) + return real_load_expr(*args, **kwargs) + + # Two cold builders also race for the content-addressed clone of the source (``ensure_cas_path`` writes a fixed + # ``.parquet.tmp``), which is a second instance of the same defect but not the one injected below. + # Building any other entry over the same source first leaves the clone in place, so that only the injection differs. + build_and_persist(project, _cheap_code(project)) + monkeypatch.setattr(compiler, "load_expr", instrumented) + start = threading.Barrier(2) + results: list = [] + errors: list[BaseException] = [] + + def builder() -> None: + try: + start.wait(timeout=10) + results.append(build_and_persist(project, _agg_code(project))) + except BaseException as exc: # noqa: BLE001 - reported by the assertion below + errors.append(exc) + + threads = _run_in_threads([builder, builder]) + + assert not any(t.is_alive() for t in threads), "a build never finished" + assert errors == [], f"a build failed: {[f'{type(e).__name__}: {e}' for e in errors]}" + assert len(results) == 2 + assert results[0].content_hash == results[1].content_hash + entry = entry_dir(project, results[0].content_hash) + for name in ("manifest.json", "schema.json", "expr.py", "xorq_build/expr.yaml"): + assert (entry / name).exists(), f"the entry directory lost {name}" + assert read_manifest(entry).content_hash == results[0].content_hash + + +def _materialize(project: str, content_hash: str): + from tallyman_xorq.materialize import materialize + + return materialize(project, content_hash) + + +def _snapshot_path(project: str, content_hash: str): + from tallyman_xorq.materialize import snapshot_path + + return snapshot_path(project, content_hash) + + +def test_concurrent_materializations_of_one_entry_all_succeed(project, orders_parquet): + """ADR-007 D11 and D4 (one writer, used by the build and by every heal): four threads materialize the same entry. + The writer takes a unique temp name in the destination directory and renames it into place, under the project + lock, so each call succeeds and the file is whole afterwards. xorq's writer used a fixed ``.parquet.tmp``: + four concurrent cold writers of one key gave three ``FileNotFoundError`` and a corrupt file that then counted as a + permanent cache hit.""" + import pyarrow.parquet as pq + + from tallyman_xorq.result_cache import snapshot_file_digest + + h = build_and_persist(project, _agg_code(project)).content_hash + manifest = read_manifest(entry_dir(project, h)) + start = threading.Barrier(4) + digests: list[str] = [] + errors: list[BaseException] = [] + + def writer() -> None: + try: + start.wait(timeout=10) + digests.append(_materialize(project, h).digest) + except BaseException as exc: # noqa: BLE001 - reported by the assertion below + errors.append(exc) + + threads = _run_in_threads([writer] * 4) + + assert not any(t.is_alive() for t in threads), "a materialization never finished" + assert errors == [], f"a materialization failed: {[f'{type(e).__name__}: {e}' for e in errors]}" + assert digests == [manifest.result_digest] * 4 + path = _snapshot_path(project, h) + assert pq.read_table(path).num_rows == manifest.row_count + assert snapshot_file_digest(path) == manifest.result_digest diff --git a/tests/test_reset_keeps_compute_cache.py b/tests/test_reset_keeps_compute_cache.py new file mode 100644 index 0000000..b18de43 --- /dev/null +++ b/tests/test_reset_keeps_compute_cache.py @@ -0,0 +1,159 @@ +"""A reset leaves ``compute_cache/`` alone, and parks clones in the bullpen instead of deleting them. + +Red tests for ``plans/ADR-007-tallyman-owned-materialization.md`` D14 (a reset leaves ``compute_cache/`` alone) and D13 +(a file is cache only if ``ensure_materialized`` can re-create it; clones are data). The scenario is +``scripts/spike_reset_roundtrip.py``, which played it through on today's code: + +- step s1 holds an entry over ``orders.parquet``; +- step s2 adds a cheap entry and a worthy entry over ``extra.parquet``, which nothing touches afterwards. + +A reset to s1 and then forward to s2 used to leave the cheap entry unreadable (``At least one path is required``), +because the backward reset deleted the content-addressed clone of ``extra.parquet`` under ``data/.cas/`` (``gc_cas`` +unlinks a clone that no surviving entry refers to) and the forward reset restores entry directories, not clones. +Snapshots under ``compute_cache/`` were pruned and copied back by a second, non-atomic writer, driven by the +git-tracked list ``compute_cache.jsonl``; the design retires that list and lets a snapshot that is missing be healed +and verified like any other (ADR-007 D5). +""" + +from __future__ import annotations + +import shutil +from pathlib import Path +from types import SimpleNamespace + +import pandas as pd +import pytest + +from tallyman_core import catalog, data_dir, read_manifest +from tallyman_core import catalog_state as cs +from tallyman_core.paths import bullpen_dir, catalog_dir, compute_cache_dir, entry_dir +from tallyman_xorq import build_and_persist +from tallyman_xorq.result_cache import cached_result_expr + + +def _recipe(project: str, source: str, tail: str = "") -> str: + return ( + "from tallyman_xorq.io import read_project_file\n" + f"t = read_project_file({source!r}, project={project!r})\n" + f"expr = t{tail}\n" + ) + + +def _stat_map(root: Path) -> dict[str, tuple[int, int, int]]: + """Every file under *root* with the fields a move, a copy or a rewrite would change.""" + out = {} + for p in sorted(root.rglob("*")): + if p.is_file(): + st = p.stat() + out[str(p.relative_to(root))] = (st.st_ino, st.st_mtime_ns, st.st_size) + return out + + +@pytest.fixture +def two_steps(project: str, orders_parquet: Path) -> SimpleNamespace: + cs.ensure_catalog_repo(project) + extra = data_dir(project) / "extra.parquet" + pd.DataFrame({"k": ["x", "y", "x", "z"], "v": [1.0, 2.0, 3.0, 4.0]}).to_parquet(extra) + + base = build_and_persist(project, _recipe(project, "orders.parquet")).content_hash + s1 = cs.checkpoint_catalog(project, "s1") + cheap = build_and_persist(project, _recipe(project, "extra.parquet", ".filter(t.v > 1)")).content_hash + worthy = build_and_persist( + project, _recipe(project, "extra.parquet", ".group_by('k').aggregate(s=t.v.sum())") + ).content_hash + s2 = cs.checkpoint_catalog(project, "s2") + assert s1 is not None and s2 is not None and s1 != s2 + + return SimpleNamespace( + project=project, + base=base, + cheap=cheap, + worthy=worthy, + s1=s1, + s2=s2, + extra_digest=read_manifest(entry_dir(project, cheap)).sources["extra.parquet"], + orders_digest=read_manifest(entry_dir(project, base)).sources["orders.parquet"], + ) + + +def _read_all(project: str, hashes: dict[str, str]) -> None: + for label, h in hashes.items(): + cached_result_expr.cache_clear() + assert len(cached_result_expr(project, h).execute()) > 0, f"the {label} entry read no rows" + + +def test_every_entry_of_the_restored_step_reads_after_a_reset_back_and_forward(two_steps): + """ADR-007 D13 and D14: after a reset to s1 and forward to s2, every entry of s2 reads, cheap and worthy, and + still reads once ``compute_cache/`` is emptied, which is the cold state of ADR-007 D7. Measured on today's code + (``scripts/spike_reset_roundtrip.py``): the cheap entry fails at once with ``At least one path is required``, and + the worthy entry fails the same way as soon as it has to heal.""" + p = two_steps.project + entries = {"cheap": two_steps.cheap, "worthy": two_steps.worthy} + + cs.reset_to(p, two_steps.s1) + cs.reset_to(p, two_steps.s2) + _read_all(p, entries) + + shutil.rmtree(compute_cache_dir(p), ignore_errors=True) + _read_all(p, entries) + + +def test_a_reset_moves_and_copies_nothing_under_compute_cache(two_steps): + """ADR-007 D14: a reset stops managing ``compute_cache/``. Snapshots are named by content hash, so a file left + behind by a retired entry cannot be served for another entry; it is unreferenced disk until that entry comes back + or the user deletes it. Nothing under the directory moves on the way back, and nothing is copied in on the way + forward (the copy was a second writer of snapshot files, outside the one writer of ADR-007 D4).""" + p = two_steps.project + root = compute_cache_dir(p) + before = _stat_map(root) + assert before, "the worthy entries of the scenario left no file under compute_cache/" + + cs.reset_to(p, two_steps.s1) + assert _stat_map(root) == before, "the reset back moved or rewrote files under compute_cache/" + + cs.reset_to(p, two_steps.s2) + assert _stat_map(root) == before, "the reset forward copied files into compute_cache/" + + +def test_a_backward_reset_parks_the_clone_in_the_bullpen_and_a_forward_reset_restores_it(two_steps): + """ADR-007 D13 and D14: a clone is the only frozen copy of the bytes an entry was built from, and it cannot be made + again once the live file has been edited. So a reset does not unlink a clone that no surviving entry refers to; it + moves it to the bullpen, as it does with entry directories, and a reset forward copies it back to ``data/.cas/``. + A clone that a surviving entry still refers to stays where it is.""" + p = two_steps.project + cas = data_dir(p) / ".cas" + extra_clone = cas / f"{two_steps.extra_digest}.parquet" + orders_clone = cas / f"{two_steps.orders_digest}.parquet" + assert extra_clone.is_file() and orders_clone.is_file() + + cs.reset_to(p, two_steps.s1) + + assert not extra_clone.exists() + assert (bullpen_dir(p) / "cas" / extra_clone.name).is_file(), "the clone was deleted, not parked in the bullpen" + assert orders_clone.is_file(), "a clone that an entry of the step still refers to must not be parked" + + cs.reset_to(p, two_steps.s2) + + assert extra_clone.is_file(), "the reset forward did not restore the clone" + + +def test_no_compute_cache_pointer_file_is_written(project: str, orders_parquet: Path): + """ADR-007 D14: ``compute_cache.jsonl``, the git-tracked list of every file that was under the directory at each + checkpoint, is retired. Capturing it was a cost every checkpoint paid, and it grew with the cache (#22).""" + cs.ensure_catalog_repo(project) + build_and_persist(project, _recipe(project, "orders.parquet", ".group_by('region').aggregate(n=t.count())")) + + assert cs.checkpoint_catalog(project, "one") is not None + + assert not (catalog_dir(project) / "compute_cache.jsonl").exists() + + +def test_the_reset_machinery_for_compute_cache_is_gone(): + """ADR-007 D14: ``prune_compute_cache`` and the ``compute_cache/`` half of ``restore_from_bullpen`` are deleted.""" + assert not hasattr(cs, "prune_compute_cache") + + +def test_compute_cache_pointer_file_is_not_on_the_tracked_surface(): + """ADR-007 D14: nothing writes ``compute_cache.jsonl`` any more, so the catalog's allowlist of tracked paths no + longer names it.""" + assert "compute_cache.jsonl" not in catalog.TRACKED_SURFACE diff --git a/tests/test_row_order.py b/tests/test_row_order.py new file mode 100644 index 0000000..d840314 --- /dev/null +++ b/tests/test_row_order.py @@ -0,0 +1,527 @@ +"""ADR-008 (plans/ADR-008-row-order-of-reads.md): every file tallyman writes carries ``__row_order``. + +Red tests for what a build produces: the column itself (ADR-008 D2), the build error for a cheap entry that drops it +(ADR-008 D3), the test that decides which entries are cheap (ADR-008 D4), the reserved name and joins (ADR-008 D6), +CSV roots (ADR-008 D7) and raw parquet reads (ADR-008 D12). Pages (ADR-008 D5) are in +``tests/test_row_order_pages.py``; sort grafting and hoisting (ADR-008 D10 and D11) in +``tests/test_row_order_sorts.py``. + +The new API these tests touch (``tallyman_xorq.materialize.snapshot_path`` and +``tallyman_xorq.worthiness.classify_expr``) is imported inside helpers, not at module top: a top-level import of a +module that does not exist yet makes ruff mis-group the block, the CI lint job fails, and the tests never run. +""" + +from __future__ import annotations + +import os +import re +from pathlib import Path + +import pyarrow as pa +import pyarrow.parquet as pq +import pytest +import xorq.api as xo +import xorq.vendor.ibis as ibis + +from tallyman_companion.diff import build_compare_expr, build_diff_expr +from tallyman_core import data_dir, read_manifest, set_alias +from tallyman_core.paths import compute_cache_dir, entry_build_dir, entry_dir +from tallyman_xorq.build import BuildError, build_and_persist +from tallyman_xorq.primary_key import resolve_primary_key +from tallyman_xorq.result_cache import cache_worthy, cached_result_expr + +ROW_ORDER = "__row_order" +ROW_ORDER_RIGHT = "__row_order_right" + +_PRELUDE = """ +import xorq.api as xo +import xorq.vendor.ibis as ibis +from tallyman_xorq.io import pinned_expr_from_alias, read_project_file, tallyman_read_csv, tracked_expr_from_alias +""" + +# A select of these columns is written the way the ADR-008 D3 error shows it: t.select("g", "n", "__row_order"). +_CORRECTED_SELECT = re.compile(r"""select\(\s*["']region["'],\s*["']price["'],\s*["']__row_order["']\s*\)""") + + +# --------------------------------------------------------------------------- # +# helpers +# --------------------------------------------------------------------------- # +def _over(project: str, source: str, body: str) -> str: + """A recipe over one source file: ``t`` is ``read_project_file(source)`` and ``expr`` is *body*.""" + return f"{_PRELUDE}t = read_project_file({source!r}, project={project!r})\nexpr = {body}\n" + + +def _orders(project: str, body: str) -> str: + """A recipe over the shoe-orders source: ``t`` is a read of orders.parquet and ``expr`` is *body*.""" + return _over(project, "orders.parquet", body) + + +def _chained(alias: str, body: str) -> str: + """A recipe that builds on the entry named *alias*: ``t`` is that entry and ``expr`` is *body*.""" + return f"{_PRELUDE}t = tracked_expr_from_alias({alias!r})\nexpr = {body}\n" + + +def _create(project: str, alias: str, code: str) -> str: + """Build *code* and name the entry, so a later recipe can chain off it; returns the content hash.""" + content_hash = build_and_persist(project, code).content_hash + set_alias(project, alias, content_hash, expect_exists=False) + return content_hash + + +def _write(project: str, name: str, columns: dict) -> Path: + """A small parquet source under the project's data dir.""" + path = data_dir(project) / name + pq.write_table(pa.table(columns), path) + return path + + +def _names(res) -> list[str]: + """The column names a build recorded in its schema.""" + return [f["name"] for f in res.schema["fields"]] + + +def _snapshot_path(project: str, content_hash: str) -> Path: + from tallyman_xorq.materialize import snapshot_path + + return snapshot_path(project, content_hash) + + +def _classify(expr): + from tallyman_xorq.worthiness import classify_expr + + return classify_expr(expr) + + +def _ordered_copies(project: str) -> list[Path]: + """Ordered copies of sources: ADR-007 D13 puts them under the project's compute cache.""" + return sorted((compute_cache_dir(project) / "ordered_sources").glob("*.parquet")) + + +def _csv_recipe(csv: Path) -> str: + return ( + "import xorq.vendor.ibis as ibis\n" + "from tallyman_xorq.io import tallyman_read_csv\n" + f"expr = tallyman_read_csv({str(csv)!r}, schema=ibis.schema({{'k': 'int64', 'v': 'int64'}}))\n" + ) + + +# --------------------------------------------------------------------------- # +# ADR-008 D2: every file tallyman reads carries __row_order +# --------------------------------------------------------------------------- # +def test_editing_a_csv_forks_the_content_hash(project): + """ADR-008 D2 (ordered copy built from the content-addressed clone), #168: an edit is a new entry. + + Today the intermediate is keyed by the CSV's path and overwritten in place, so the recipe hashes to the same + entry before and after the edit and the new rows are never seen. + """ + csv = data_dir(project) / "edited.csv" + csv.write_text("k,v\n1,10\n2,20\n3,30\n") + first = build_and_persist(project, _csv_recipe(csv)).content_hash + + csv.write_text("k,v\n1,10\n2,999\n3,30\n4,40\n") + st = csv.stat() + os.utime(csv, ns=(st.st_atime_ns, st.st_mtime_ns + 5_000_000_000)) # newer than any filesystem clock granularity + second = build_and_persist(project, _csv_recipe(csv)).content_hash + + assert second != first, "editing the CSV and re-running the same recipe must create a new entry" + cached_result_expr.cache_clear() + assert cached_result_expr(project, first).execute()["v"].tolist() == [10, 20, 30] + assert cached_result_expr(project, second).execute()["v"].tolist() == [10, 999, 30, 40] + + +def test_a_worthy_snapshot_ends_in_row_order(project, orders_parquet): + """ADR-008 D2: a snapshot ends in an int64 ``__row_order`` holding ``0..N-1`` in the file's row order.""" + res = build_and_persist(project, _orders(project, "t.order_by(t.price.desc())")) + assert _names(res)[-1] == ROW_ORDER, f"the entry's schema must end in __row_order, got {_names(res)}" + + table = pq.read_table(_snapshot_path(project, res.content_hash)) + assert table.column_names[-1] == ROW_ORDER + assert table.schema.field(ROW_ORDER).type == pa.int64() + assert table[ROW_ORDER].to_pylist() == list(range(table.num_rows)) + prices = table["price"].to_pylist() + assert prices == sorted(prices, reverse=True), "numbered in the order the file is written in" + + +def test_the_ordered_copy_of_a_parquet_source_ends_in_row_order(project, orders_parquet): + """ADR-008 D2: a parquet source enters tallyman through an ordered copy with ``__row_order`` last.""" + build_and_persist(project, _orders(project, "t")) + + copies = _ordered_copies(project) + assert len(copies) == 1, f"expected one ordered copy under compute_cache/ordered_sources, found {copies}" + copy, source = pq.read_table(copies[0]), pq.read_table(orders_parquet) + assert copy.column_names == [*source.column_names, ROW_ORDER] + assert copy.schema.field(ROW_ORDER).type == pa.int64() + assert copy[ROW_ORDER].to_pylist() == list(range(source.num_rows)) + for name in source.column_names: + assert copy[name].to_pylist() == source[name].to_pylist(), f"{name} must be copied in file order" + + +def test_the_ordered_copy_of_a_csv_source_ends_in_row_order(project): + """ADR-008 D2: a CSV source enters tallyman through an ordered copy with ``__row_order`` last.""" + csv = data_dir(project) / "ordered.csv" + csv.write_text("k,v\n3,30\n1,10\n2,20\n") + build_and_persist(project, _csv_recipe(csv)) + + copies = _ordered_copies(project) + assert len(copies) == 1, f"expected one ordered copy under compute_cache/ordered_sources, found {copies}" + copy = pq.read_table(copies[0]) + assert copy.column_names == ["k", "v", ROW_ORDER] + assert copy["k"].to_pylist() == [3, 1, 2] # file order, not sorted order + assert copy[ROW_ORDER].to_pylist() == [0, 1, 2] + + +def test_a_parquet_source_with_its_own_row_order_column_has_it_overwritten(project): + """ADR-008 D2: a source that already has ``__row_order`` (a file tallyman exported) has it overwritten.""" + _write(project, "exported.parquet", {"k": [10, 20, 30, 40], ROW_ORDER: [5, 3, 9, 1]}) + code = f"{_PRELUDE}expr = read_project_file('exported.parquet', project={project!r})\n" + res = build_and_persist(project, code) + + df = cached_result_expr(project, res.content_hash).execute() + assert list(df.columns) == ["k", ROW_ORDER] + assert df["k"].tolist() == [10, 20, 30, 40] + assert df[ROW_ORDER].tolist() == [0, 1, 2, 3], "the source's own values must be overwritten with 0..N-1" + + +def test_a_worthy_entry_that_keeps_its_parents_rows_renumbers_them(project, orders_parquet): + """ADR-008 D2: ``materialize`` replaces an inherited ``__row_order`` with positions in its own file. + + A filter leaves gaps in the parent's positions. The entry is worthy because of the window function, and its file + numbers its own rows ``0..M-1`` with no gaps. + """ + _create(project, "orders", _orders(project, "t")) + child = build_and_persist(project, _chained("orders", "t.filter(t.qty > 2).mutate(rn=ibis.row_number())")) + assert _names(child)[-1] == ROW_ORDER, f"the entry's schema must end in __row_order, got {_names(child)}" + + table = pq.read_table(_snapshot_path(project, child.content_hash)) + assert 0 < table.num_rows < 200 + assert table.column_names[-1] == ROW_ORDER + assert table[ROW_ORDER].to_pylist() == list(range(table.num_rows)) + + +# --------------------------------------------------------------------------- # +# ADR-008 D3: a cheap entry that drops __row_order is a build error +# --------------------------------------------------------------------------- # +def test_a_cheap_select_that_omits_row_order_names_its_parent_and_the_fix(project, orders_parquet): + """ADR-008 D3: the error names the parent entry and shows the corrected select.""" + parent = _create(project, "orders", _orders(project, "t")) + with pytest.raises(BuildError) as exc: + build_and_persist(project, _chained("orders", "t.select('region', 'price')")) + msg = str(exc.value) + assert ROW_ORDER in msg, msg + assert _CORRECTED_SELECT.search(msg), f"the message must show the corrected select: {msg}" + assert parent in msg or parent[:12] in msg or "orders" in msg, f"the message must name the parent entry: {msg}" + + +def test_a_cheap_select_over_a_source_read_omits_row_order_with_the_fix(project, orders_parquet): + """ADR-008 D3: the same error when the cheap entry reads a source file directly.""" + with pytest.raises(BuildError) as exc: + build_and_persist(project, _orders(project, "t.select('region', 'price')")) + msg = str(exc.value) + assert ROW_ORDER in msg, msg + assert _CORRECTED_SELECT.search(msg), f"the message must show the corrected select: {msg}" + + +def test_the_same_select_over_a_worthy_recipe_builds_and_is_numbered(project, orders_parquet): + """ADR-008 D3: a worthy entry is exempt, because the writer numbers its rows.""" + res = build_and_persist( + project, _orders(project, "t.group_by('region').aggregate(n=t.count()).select('region', 'n')") + ) + assert _names(res) == ["region", "n", ROW_ORDER] + + +def test_a_computed_column_added_after_row_order_leaves_it_last(project, orders_parquet): + """ADR-008 D3: tallyman moves ``__row_order`` to the last position, at the top of the expression only.""" + res = build_and_persist(project, _orders(project, "t.mutate(double=t.price * 2)")) + names = _names(res) + assert "double" in names + assert names.count(ROW_ORDER) == 1 and names[-1] == ROW_ORDER, names + assert cache_worthy(project, res.content_hash) is False + + +# --------------------------------------------------------------------------- # +# ADR-008 D3 / D6: asking for an order renumbers, assigning is an error +# --------------------------------------------------------------------------- # +def test_asking_for_an_order_numbers_the_file_in_that_order(project): + """ADR-008 D3: an ``order_by`` makes the entry worthy and the writer numbers the rows in the requested order.""" + _write(project, "amounts.parquet", {"name": list("abcdef"), "amount": [40, 10, 60, 20, 50, 30]}) + res = build_and_persist(project, _over(project, "amounts.parquet", "t.order_by(t.amount.desc())")) + + df = cached_result_expr(project, res.content_hash).execute() + assert ROW_ORDER in df.columns, f"the entry must carry __row_order, got {list(df.columns)}" + assert df["amount"].tolist() == [60, 50, 40, 30, 20, 10] + assert df[ROW_ORDER].tolist() == [0, 1, 2, 3, 4, 5] + + +@pytest.mark.parametrize( + "body", + [ + pytest.param("t.mutate(__row_order=t.order_id * 2)", id="mutate"), + pytest.param("t.select('region', __row_order=t.order_id)", id="select"), + pytest.param("t.mutate(__row_order=t.order_id).order_by('region')", id="worthy entry"), + ], +) +def test_assigning_to_row_order_is_a_build_error(project, orders_parquet, body): + """ADR-008 D6: arbitrary values could contain ties or gaps, so a recipe may not assign to the column.""" + with pytest.raises(BuildError) as exc: + build_and_persist(project, _orders(project, body)) + assert ROW_ORDER in str(exc.value) + + +def test_a_debugging_copy_of_row_order_survives_materialization(project, orders_parquet): + """ADR-008 D6: ``__row_order_v1`` is ordinary data: it keeps the parent's positions after the child is written.""" + parent = _create(project, "foo_v1", _orders(project, "t.order_by(t.price.desc())")) + child_code = ( + f"{_PRELUDE}t = tracked_expr_from_alias('foo_v1')\n" + "c = t.mutate(__row_order_v1=t['__row_order'])\n" + "expr = c.order_by(c.qty)\n" + ) + child = build_and_persist(project, child_code) + + parent_df = pq.read_table(_snapshot_path(project, parent)).to_pandas() + kid = pq.read_table(_snapshot_path(project, child.content_hash)).to_pandas() + assert "__row_order_v1" in kid.columns and list(kid.columns)[-1] == ROW_ORDER + assert kid[ROW_ORDER].tolist() == list(range(len(kid))), "the child's own positions" + joined = kid.merge(parent_df[["order_id", ROW_ORDER]], on="order_id", suffixes=("", "_parent")) + assert len(joined) == len(kid) + assert (joined["__row_order_v1"] == joined[f"{ROW_ORDER}_parent"]).all(), "the copy keeps the parent's positions" + assert not (joined[ROW_ORDER] == joined[f"{ROW_ORDER}_parent"]).all(), "the sort by qty moved the rows" + + +# --------------------------------------------------------------------------- # +# ADR-008 D4: cheap means row-preserving over one file +# --------------------------------------------------------------------------- # +def _live_tables(tmp_path: Path): + """Two files that already carry ``__row_order``, read as live expressions (nothing is built).""" + n = 6 + tags = [["x", "y"], ["z"], [], ["x"], ["y", "z", "w"], ["q"]] + pq.write_table( + pa.table({"k": list(range(n)), "v": [1.5, 2.5, None, 4.5, 5.5, 6.5], "tags": tags, ROW_ORDER: list(range(n))}), + tmp_path / "t.parquet", + ) + pq.write_table(pa.table({"k": [1, 3, 5], ROW_ORDER: [0, 1, 2]}), tmp_path / "u.parquet") + return xo.deferred_read_parquet(str(tmp_path / "t.parquet")), xo.deferred_read_parquet(str(tmp_path / "u.parquet")) + + +_WORTHY_SHAPES = { + "union": lambda t, u: t.union(t), + "distinct": lambda t, u: t.select("k", ROW_ORDER).distinct(), + "relation-level unnest": lambda t, u: t.unnest("tags"), + "an operation the allow-list has never seen": lambda t, u: t.sample(0.5), + "limit": lambda t, u: t.limit(3), + "sort": lambda t, u: t.order_by("k"), + "aggregate": lambda t, u: t.group_by("k").aggregate(n=t.count()), + "join": lambda t, u: t.join(u, "k"), + "value-level unnest inside a select": lambda t, u: t.select("k", ROW_ORDER, tag=t.tags.unnest()), + "row_number() in a mutate": lambda t, u: t.mutate(rn=ibis.row_number()), + "lag() in a mutate": lambda t, u: t.mutate(prev=t.v.lag()), + "random() in a mutate": lambda t, u: t.mutate(r=ibis.random()), + "now() in a mutate": lambda t, u: t.mutate(ts=ibis.now()), + "a filter against a second file": lambda t, u: t.filter(t.k.isin(u.k)), +} +_CHEAP_SHAPES = { + "a bare read": lambda t, u: t, + "filter and a computed column": lambda t, u: t.filter(t.k > 0).mutate(w=t.v * 2), + "rename, cast and drop": lambda t, u: t.rename(key="k").mutate(v=t.v.cast("float32")).drop("tags"), + "drop_null": lambda t, u: t.drop_null(["v"]), + "fill_null": lambda t, u: t.fill_null({"v": 0.0}), +} + + +@pytest.mark.parametrize("label", list(_WORTHY_SHAPES)) +def test_the_cheap_test_classifies_these_shapes_as_worthy(tmp_path, label): + """ADR-008 D4: an unknown operation, a second file, and value operations that multiply rows or read the order.""" + t, u = _live_tables(tmp_path) + verdict = _classify(_WORTHY_SHAPES[label](t, u)) + assert verdict.worthy is True, f"{label} must be worthy: {verdict}" + + +@pytest.mark.parametrize("label", list(_CHEAP_SHAPES)) +def test_the_cheap_test_classifies_these_shapes_as_cheap(tmp_path, label): + """ADR-008 D4: a file read, filter, selection, computed column, rename, cast, drop, drop_null, fill_null.""" + t, u = _live_tables(tmp_path) + verdict = _classify(_CHEAP_SHAPES[label](t, u)) + assert verdict.worthy is False, f"{label} must be cheap: {verdict}" + + +def test_the_verdict_is_read_from_the_manifest_with_no_expr_yaml_parsed(project, orders_parquet, monkeypatch): + """ADR-008 D4: computed once at build and recorded; ``classify_build`` and its regex over expr.yaml are retired.""" + cheap = build_and_persist(project, _orders(project, "t.filter(t.qty > 1).mutate(double=t.price * 2)")).content_hash + worthy = build_and_persist( + project, _orders(project, "t.group_by('region').aggregate(total=t.price.sum())") + ).content_hash + assert read_manifest(entry_dir(project, cheap)).cache_worthy is False + assert read_manifest(entry_dir(project, worthy)).cache_worthy is True + + real_read_text = Path.read_text + + def guarded(self, *args, **kwargs): + if self.suffix in {".yaml", ".yml"}: + raise AssertionError(f"the worthiness verdict parsed {self}") + return real_read_text(self, *args, **kwargs) + + monkeypatch.setattr(Path, "read_text", guarded) + assert cache_worthy(project, cheap) is False + assert cache_worthy(project, worthy) is True + + +def test_a_value_level_unnest_makes_the_entry_worthy(project): + """ADR-008 D4: an ``unnest`` inside a select multiplies rows, which a list of relation operations cannot see.""" + _write(project, "lists.parquet", {"k": [1, 2, 3], "tags": [["x", "y"], ["z"], ["x", "y", "z"]]}) + res = build_and_persist(project, _over(project, "lists.parquet", "t.select('k', tag=t.tags.unnest())")) + assert res.cache_worthy is True + assert read_manifest(entry_dir(project, res.content_hash)).cache_worthy is True + + +@pytest.mark.parametrize( + "body", + [ + pytest.param("t.union(t)", id="union"), + pytest.param("t.select('region', 'category').distinct()", id="distinct"), + ], +) +def test_a_union_and_a_distinct_are_materialized(project, orders_parquet, body): + """ADR-008 D4: neither can carry one parent's row order, and today's deny-list classes both as cheap.""" + res = build_and_persist(project, _orders(project, body)) + assert res.cache_worthy is True + assert _names(res)[-1] == ROW_ORDER + + +# --------------------------------------------------------------------------- # +# ADR-008 D6: the reserved name +# --------------------------------------------------------------------------- # +def test_the_primary_key_search_never_returns_row_order(project, orders_parquet): + """ADR-008 D6: ``__row_order`` is unique in every table, so it would win the search for any table without a key.""" + h = build_and_persist(project, _orders(project, "t.select('region', 'category').order_by('region')")).content_hash + assert ROW_ORDER in cached_result_expr(project, h).columns, "the entry must carry __row_order for this test" + + key = resolve_primary_key(project, h) + assert ROW_ORDER not in key + assert key == [], "region and category alone are not unique, and __row_order does not count" + + +def test_a_diff_carries_no_row_order_column_from_either_side(project, orders_parquet): + """ADR-008 D6: ``build_compare_expr`` and ``build_diff_expr`` drop it from both sides before joining.""" + a = _create(project, "orders", _orders(project, "t")) + b = build_and_persist(project, _chained("orders", "t.filter(t.qty > 1)")).content_hash + a_expr, b_expr = cached_result_expr(project, a), cached_result_expr(project, b) + assert ROW_ORDER in a_expr.columns and ROW_ORDER in b_expr.columns, "both inputs must carry __row_order" + + compared, _overrides = build_compare_expr(a_expr, b_expr, ["order_id"]) + assert not [c for c in compared.columns if c.startswith(ROW_ORDER)], list(compared.columns) + promoted = build_diff_expr(a, b, ["order_id"]) + assert not [c for c in promoted.columns if c.startswith(ROW_ORDER)], list(promoted.columns) + + +# --------------------------------------------------------------------------- # +# ADR-008 D6: joins +# --------------------------------------------------------------------------- # +def _keyed_sources(project: str) -> None: + for name in ("a", "b", "c"): + _write(project, f"{name}.parquet", {"k": list(range(5)), name: [f"{name}{i}" for i in range(5)]}) + + +def _three_way(project: str, right_side: str = "{}") -> str: + """Three sources joined in one recipe; *right_side* is a template for how the right-hand inputs are written.""" + return ( + f"{_PRELUDE}" + f"a = read_project_file('a.parquet', project={project!r})\n" + f"b = read_project_file('b.parquet', project={project!r})\n" + f"c = read_project_file('c.parquet', project={project!r})\n" + f"expr = a.join({right_side.format('b')}, 'k').join({right_side.format('c')}, 'k')\n" + ) + + +def test_a_join_entrys_file_has_no_right_hand_row_order(project): + """ADR-008 D6: ibis renames the right side's copy to ``__row_order_right``; the writer drops it.""" + _keyed_sources(project) + code = ( + f"{_PRELUDE}a = read_project_file('a.parquet', project={project!r})\n" + f"b = read_project_file('b.parquet', project={project!r})\nexpr = a.join(b, 'k')\n" + ) + res = build_and_persist(project, code) + assert _names(res)[-1] == ROW_ORDER, _names(res) + + names = pq.read_table(_snapshot_path(project, res.content_hash)).column_names + assert ROW_ORDER_RIGHT not in names + assert names[-1] == ROW_ORDER + assert sorted(names) == sorted(["k", "a", "b", ROW_ORDER]) + + +def test_a_join_entry_can_be_joined_to_a_third_entry(project): + """ADR-008 D6: with the right-hand copy dropped from its file, the join entry joins to a third entry.""" + _keyed_sources(project) + ab = ( + f"{_PRELUDE}a = read_project_file('a.parquet', project={project!r})\n" + f"b = read_project_file('b.parquet', project={project!r})\nexpr = a.join(b, 'k')\n" + ) + _create(project, "ab", ab) + third = ( + f"{_PRELUDE}ab = tracked_expr_from_alias('ab')\n" + f"c = read_project_file('c.parquet', project={project!r})\nexpr = ab.join(c, 'k')\n" + ) + res = build_and_persist(project, third) + + names = pq.read_table(_snapshot_path(project, res.content_hash)).column_names + assert ROW_ORDER_RIGHT not in names + assert names[-1] == ROW_ORDER + assert sorted(names) == sorted(["k", "a", "b", "c", ROW_ORDER]) + + +def test_a_three_way_join_in_one_recipe_tells_the_author_to_drop_the_right_hand_copies(project): + """ADR-008 D6: ibis fails on ``__row_order_right``, a name the author never wrote, so the build says what to do.""" + _keyed_sources(project) + with pytest.raises(BuildError) as exc: + build_and_persist(project, _three_way(project)) + msg = str(exc.value) + assert re.search(r"""drop\(\s*["']__row_order["']\s*\)""", msg), f"the message must say to drop it: {msg}" + assert "right" in msg.lower(), msg + + +def test_a_three_way_join_builds_when_the_right_hand_inputs_drop_row_order(project): + """ADR-008 D6: the instruction the error gives works.""" + _keyed_sources(project) + res = build_and_persist(project, _three_way(project, "{}.drop('__row_order')")) + names = pq.read_table(_snapshot_path(project, res.content_hash)).column_names + assert ROW_ORDER_RIGHT not in names + assert names[-1] == ROW_ORDER + assert sorted(names) == sorted(["k", "a", "b", "c", ROW_ORDER]) + + +# --------------------------------------------------------------------------- # +# ADR-008 D7: CSV roots +# --------------------------------------------------------------------------- # +def test_a_csv_root_is_a_cheap_read_with_exactly_one_row_order_column(project): + """ADR-008 D7: ``tallyman_read_csv`` loses its trailing ``order_by`` and its column becomes ``__row_order``.""" + csv = data_dir(project) / "root.csv" + csv.write_text("k,v\n3,30\n1,10\n2,20\n") + res = build_and_persist(project, _csv_recipe(csv)) + + yaml_text = (entry_build_dir(project, res.content_hash) / "expr.yaml").read_text() + assert not re.search(r"op:\s*Sort\b", yaml_text), "a CSV root is a plain read of its ordered copy: no Sort" + assert cache_worthy(project, res.content_hash) is False + assert read_manifest(entry_dir(project, res.content_hash)).cache_worthy is False + + names = _names(res) + assert names == ["k", "v", ROW_ORDER], names + assert "original_row_order" not in names + + +# --------------------------------------------------------------------------- # +# ADR-008 D12: a raw parquet read is a build error +# --------------------------------------------------------------------------- # +def test_a_raw_parquet_read_of_a_source_file_is_a_build_error(project, orders_parquet): + """ADR-008 D12: such a read has no digest, no clone and no ordered copy, so no ``__row_order``.""" + code = f"import xorq.api as xo\nexpr = xo.deferred_read_parquet({str(orders_parquet)!r})\n" + with pytest.raises(BuildError) as exc: + build_and_persist(project, code) + assert "read_project_file" in str(exc.value) + + +def test_a_raw_parquet_read_of_a_file_outside_the_project_is_a_build_error(project, tmp_path): + """ADR-008 D12: only tallyman's own files (snapshots and ordered copies under compute_cache) may be read raw.""" + outside = tmp_path / "elsewhere.parquet" + pq.write_table(pa.table({"k": [1, 2, 3]}), outside) + code = f"import xorq.api as xo\nt = xo.deferred_read_parquet({str(outside)!r})\nexpr = t.filter(t.k > 1)\n" + with pytest.raises(BuildError) as exc: + build_and_persist(project, code) + assert "read_project_file" in str(exc.value) diff --git a/tests/test_row_order_pages.py b/tests/test_row_order_pages.py new file mode 100644 index 0000000..2190b34 --- /dev/null +++ b/tests/test_row_order_pages.py @@ -0,0 +1,91 @@ +"""ADR-008 D5 (plans/ADR-008-row-order-of-reads.md): every page request orders by ``__row_order``. + +A page is a function of ``(content_hash, sort, offset, limit)``. Today ``/api/data`` serves +``cached_result_expr(...).limit(limit, offset=offset)`` with no ``ORDER BY``, and above DataFusion's scan-split +threshold (10,485,760 bytes) an unordered ``LIMIT/OFFSET`` returns different rows on each request, so paging through a +large entry repeats some rows and never shows others. + +Asserting that eight identical requests agree with each other is not enough: one rejected engine setting returned the +same wrong page eight times out of eight (ADR-008 D5). Each response is compared with the rows that sit at those +positions in the file, computed independently with pandas from the source. +""" + +from __future__ import annotations + +import shutil +from pathlib import Path + +import pyarrow.parquet as pq +import pytest +from fastapi.testclient import TestClient + +from tallyman_core import data_dir +from tallyman_xorq.build import build_and_persist +from tests.big_parquet import write_big_parquet + +ROW_ORDER = "__row_order" +OFFSET, LIMIT, REQUESTS = 1_000_000, 50, 8 + +_PRELUDE = "from tallyman_xorq.io import read_project_file\n" + + +@pytest.fixture(scope="module") +def big_source(tmp_path_factory) -> Path: + """A 26 MB parquet file, written once for the module: id (the file position), g (200 values), v (float).""" + return write_big_parquet(tmp_path_factory.mktemp("big_source") / "big.parquet") + + +@pytest.fixture(scope="module") +def big_frame(big_source): + return pq.read_table(big_source).to_pandas() + + +@pytest.fixture +def big(project, big_source) -> Path: + dest = data_dir(project) / "big.parquet" + shutil.copyfile(big_source, dest) + return dest + + +def _recipe(project: str, body: str) -> str: + return f"{_PRELUDE}t = read_project_file('big.parquet', project={project!r})\nexpr = {body}\n" + + +def _requests(client: TestClient, project: str, content_hash: str) -> list[list[dict]]: + """Eight identical page requests at a deep offset; each response must carry the row-order column.""" + pages = [] + for i in range(REQUESTS): + r = client.get(f"/{project}/api/data/{content_hash}?offset={OFFSET}&limit={LIMIT}") + assert r.status_code == 200, r.text + rows = r.json()["data"] + assert len(rows) == LIMIT + assert ROW_ORDER in rows[0], f"request {i}: a row must end in {ROW_ORDER}, got {list(rows[0])}" + pages.append(rows) + return pages + + +@pytest.mark.parametrize( + ("body", "max_g"), + [ + pytest.param("t.order_by('id')", None, id="worthy entry"), + pytest.param("t.filter(t.g >= 0)", None, id="cheap entry that keeps every row"), + pytest.param("t.filter(t.g < 150)", 150, id="cheap entry that drops rows"), + ], +) +def test_a_deep_page_is_exactly_the_rows_at_those_positions(fresh_companion_app, project, big, big_frame, body, max_g): + """ADR-008 D5: eight identical requests each return the rows at positions offset..offset+limit-1. + + ``id`` is the file position, so for the entries that keep every row ``__row_order`` equals ``id``. A cheap entry + that drops rows inherits its parent's positions, which then have gaps: its page is the surviving rows at + ``OFFSET`` in ``__row_order`` order, and ``__row_order`` still holds each row's position in the source. + """ + h = build_and_persist(project, _recipe(project, body)).content_hash + kept = big_frame if max_g is None else big_frame[big_frame["g"] < max_g] + expected = kept["id"].iloc[OFFSET : OFFSET + LIMIT].tolist() + assert len(expected) == LIMIT, "the fixture must have rows at the requested offset" + if max_g is None: + assert expected == list(range(OFFSET, OFFSET + LIMIT)) + + for i, rows in enumerate(_requests(TestClient(fresh_companion_app), project, h)): + assert [r["id"] for r in rows] == expected, f"request {i} returned other rows than positions {OFFSET}.." + assert [r[ROW_ORDER] for r in rows] == expected, f"request {i}: __row_order must be the position in the file" diff --git a/tests/test_row_order_sorts.py b/tests/test_row_order_sorts.py new file mode 100644 index 0000000..2c46f95 --- /dev/null +++ b/tests/test_row_order_sorts.py @@ -0,0 +1,160 @@ +"""ADR-008 D10 and D11 (plans/ADR-008-row-order-of-reads.md): where a unique sort has to be imposed. + +Paddy's rule: when a supplied sort is not deterministic, the natural order (``__row_order``) is imposed into each +``order_by``. Today ``_canonical_sorted`` extends an author's sort only when it is the top node of the expression, and +wraps everything else in a sort that leads with the inherited row order, which is unique, so an author's sort that +is followed by another step has no effect on what is written. + +- ADR-008 D10: every ``Sort`` node gets ``__row_order`` and then the remaining sortable columns as its last keys, so a + sort that feeds a ``limit`` decides the same rows on any connection. +- ADR-008 D11: a sort that is not the recipe's last step is hoisted: the top-level sort leads with its keys. A key that + did not survive (dropped, overwritten, or an expression) is a build error that names it. +""" + +from __future__ import annotations + +import shutil +from pathlib import Path + +import pyarrow as pa +import pyarrow.parquet as pq +import pytest + +from tallyman_core import data_dir +from tallyman_xorq.build import BuildError, build_and_persist +from tallyman_xorq.result_cache import cached_result_expr +from tests.big_parquet import write_big_parquet + +ROW_ORDER = "__row_order" + +_PRELUDE = "import xorq.vendor.ibis as ibis\nfrom tallyman_xorq.io import read_project_file\n" + +# The parent's rows are in this order, so an entry that ignores the author's sort is written 40, 10, 60, ... +AMOUNTS = [40, 10, 60, 20, 50, 30] + + +@pytest.fixture +def amounts(project: str) -> Path: + path = data_dir(project) / "amounts.parquet" + pq.write_table(pa.table({"name": list("abcdef"), "amount": AMOUNTS}), path) + return path + + +@pytest.fixture(scope="module") +def big_source(tmp_path_factory) -> Path: + return write_big_parquet(tmp_path_factory.mktemp("big_source") / "big.parquet") + + +def _sorted_then(project: str, body: str, source: str = "amounts.parquet") -> str: + """A recipe whose ``by`` is the source ordered by amount, descending, followed by *body*.""" + return ( + f"{_PRELUDE}t = read_project_file({source!r}, project={project!r})\n" + "by = t.order_by(t.amount.desc())\n" + f"expr = {body}\n" + ) + + +def _sort_key_names(project: str, content_hash: str) -> list[list[str]]: + """The column names in the keys of every ``Sort`` node of the entry's frozen build.""" + import xorq.vendor.ibis.expr.operations as ops + from xorq.common.utils.graph_utils import walk_nodes + + from tallyman_xorq.result_cache import load_entry_expr + + loaded = load_entry_expr(project, content_hash) + return [[k.expr.name for k in sort.keys if isinstance(k.expr, ops.Field)] for sort in walk_nodes(ops.Sort, loaded)] + + +# --------------------------------------------------------------------------- # +# ADR-008 D10: the natural order is imposed on every order_by +# --------------------------------------------------------------------------- # +def test_a_sort_that_feeds_a_limit_keeps_the_rows_the_natural_order_picks(project, big_source): + """ADR-008 D10: ``order_by(g).limit(1000)`` holds exactly the rows that ``(g, __row_order)`` picks. + + About 7,500 rows tie on the smallest ``g`` and the limit cuts through them, so which 1,000 the entry holds is + decided by the tie-break of the sort that feeds the limit, not by the sort above it. ``id`` is the file position, + so ``(g, id)`` is ``(g, __row_order)``. + """ + shutil.copyfile(big_source, data_dir(project) / "big.parquet") + code = f"{_PRELUDE}t = read_project_file('big.parquet', project={project!r})\nexpr = t.order_by('g').limit(1000)\n" + res = build_and_persist(project, code) + + frame = pq.read_table(big_source).to_pandas() + expected = set(frame.sort_values(["g", "id"]).head(1000)["id"]) + got = set(cached_result_expr(project, res.content_hash).execute()["id"]) + assert len(got) == 1000 + assert got == expected, f"{len(got ^ expected)} rows differ from the 1,000 that (g, __row_order) picks" + + +def test_every_sort_in_the_graph_carries_the_row_order_tie_break(project, amounts): + """ADR-008 D10: the keys the author wrote, then ``__row_order``, at every ``Sort`` node and not only the top one.""" + res = build_and_persist(project, _sorted_then(project, "by.limit(3)")) + + sorts = _sort_key_names(project, res.content_hash) + assert sorts, "the entry's build must contain a Sort" + for keys in sorts: + assert ROW_ORDER in keys, f"a Sort without the natural-order tie-break: keys {keys}" + assert keys.index("amount") < keys.index(ROW_ORDER), f"the author's key comes before the tie-break: {keys}" + + +# --------------------------------------------------------------------------- # +# ADR-008 D11: a sort that is not the last step is hoisted, or the build fails +# --------------------------------------------------------------------------- # +@pytest.mark.parametrize( + ("body", "column", "expected"), + [ + pytest.param("by.limit(3)", "amount", [60, 50, 40], id="a top-three entry is written in rank order"), + pytest.param("by.mutate(double=by.amount * 2)", "amount", [60, 50, 40, 30, 20, 10], id="order_by then mutate"), + pytest.param("by.filter(by.amount > 15)", "amount", [60, 50, 40, 30, 20], id="order_by then filter"), + pytest.param("by.select(by.name, by.amount)", "amount", [60, 50, 40, 30, 20, 10], id="order_by then select"), + pytest.param("by.select(by.name, cost=by.amount)", "cost", [60, 50, 40, 30, 20, 10], id="a rename is followed"), + ], +) +def test_a_sort_that_is_not_the_last_step_decides_the_written_order(project, amounts, body, column, expected): + """ADR-008 D11: the author asked for amount descending, and it is kept; the file is numbered in that order.""" + res = build_and_persist(project, _sorted_then(project, body)) + + df = cached_result_expr(project, res.content_hash).execute() + assert df[column].tolist() == expected + assert df[ROW_ORDER].tolist() == list(range(len(expected))) + assert list(df.columns)[-1] == ROW_ORDER + + +def _assert_names_the_sort_key(msg: str) -> None: + """The message names the key and talks about the sort, and comes from the build, not from executing the plan. + + Overwriting the key of a sort already fails today, but only when the plan is executed, with DataFusion's + "Schema contains qualified field name t0.amount and unqualified field name amount which would be ambiguous". That + message names ``amount`` too, so naming the key is not enough to tell the two apart. + """ + assert "amount" in msg, msg + assert "execution failed" not in msg, f"the build must refuse the recipe before it executes it: {msg[:300]}" + mentions_a_sort = "sort" in msg.lower() or "order_by" in msg.lower() + assert mentions_a_sort, f"the message must say that a sort is involved: {msg[:300]}" + + +@pytest.mark.parametrize( + "body", + [ + pytest.param("by.select(by.name)", id="the key is dropped"), + pytest.param("by.mutate(amount=by.amount * -1)", id="the key is overwritten by a mutate"), + pytest.param("by.select(by.name, amount=by.amount.cast('float64'))", id="the key is redefined in a select"), + ], +) +def test_a_sort_key_that_did_not_survive_is_a_build_error_naming_it(project, amounts, body): + """ADR-008 D11: report what the author can fix in one line: keep the column, or sort as the last step.""" + with pytest.raises(BuildError) as exc: + build_and_persist(project, _sorted_then(project, body)) + _assert_names_the_sort_key(str(exc.value)) + + +def test_a_sort_key_that_was_an_expression_is_a_build_error_when_a_step_follows(project, amounts): + """ADR-008 D11: an expression is not an output column, so there is nothing to lead the top-level sort with.""" + code = ( + f"{_PRELUDE}t = read_project_file('amounts.parquet', project={project!r})\n" + "by = t.order_by(t.amount + 1)\n" + "expr = by.mutate(double=by.amount * 2)\n" + ) + with pytest.raises(BuildError) as exc: + build_and_persist(project, code) + _assert_names_the_sort_key(str(exc.value)) diff --git a/tests/test_snapshot_format.py b/tests/test_snapshot_format.py new file mode 100644 index 0000000..886405f --- /dev/null +++ b/tests/test_snapshot_format.py @@ -0,0 +1,691 @@ +"""ADR-009: the snapshot's format, single-partition materialization, and the create-twice reproducibility check. + +Decisions covered, all from ``plans/ADR-009-digest-stability.md``: + +- D1 (materialization runs single-partition): float aggregates heal to the digest they were built with, and the plan + runs on a connection with ``target_partitions = 1`` that is not the process default. +- D3 (the snapshot's format): one row group is 1,048,576 rows, zstd, format 2.6, statistics and a page index, and + ``__row_order`` is the last column. The ordered copy of a source has 122,880-row groups. ``schema.json`` is read from + the file that was written. +- D4 (a mismatch record names its likely cause): the manifest records engine versions and the snapshot format + version, and an unfaithful heal says "the engine changed" when it did. +- D6 (create runs the query twice and compares): a recipe that differs run to run is recorded as not reproducible and + its file is pinned; a heal runs the query once. + +The digest definition itself (D2) is in ``tests/test_digest.py``. + +Names that do not exist yet (``tallyman_xorq.materialize`` and the manifest and result fields the contract adds) are +imported inside the tests, so each test fails on its own when its name is missing. The tests that need a source above +DataFusion's 10,485,760-byte scan-split threshold share one 26 MB file (``big_source``) and, for the layout +assertions, one built entry (``big_built``). +""" + +from __future__ import annotations + +import json +import os +import shutil +from dataclasses import dataclass +from importlib.metadata import version +from pathlib import Path + +import numpy as np +import pyarrow as pa +import pyarrow.compute as pc +import pyarrow.parquet as pq +import pytest +from fastapi.testclient import TestClient +from xorq.config import default_backend + +from tallyman_core import data_dir, ensure_project +from tallyman_core.errors import list_errors +from tallyman_core.manifest import read_manifest +from tallyman_core.paths import compute_cache_dir, entry_dir, entry_schema_path +from tallyman_xorq.build import build_and_persist +from tallyman_xorq.result_cache import ( + UNFAITHFUL_HEAL_HOOKS, + baked_snapshot_path, + cached_result_expr, + snapshot_file_digest, + verify_result_faithful, +) +from tests.big_parquet import write_big_parquet + +PREFIX = "arrow-sha256:" +SNAPSHOT_ROW_GROUP = 1_048_576 # ADR-009 D3 +ORDERED_COPY_ROW_GROUP = 122_880 # ADR-009 D3, io._CSV_PARQUET_WRITE +BIG_ROWS = 1_500_000 # tests/big_parquet.py default + + +# --------------------------------------------------------------------------- +# helpers +# --------------------------------------------------------------------------- + + +def _materialize_module(): + import tallyman_xorq.materialize as materialize_module + + return materialize_module + + +def _show(backend, key: str) -> str: + return str(backend.raw_sql(f"SHOW {key}").to_pandas().iloc[0, 1]) + + +def _unfaithful(project: str) -> list[dict]: + return [e for e in list_errors(project, limit=1_000_000) if e.get("code") == "unfaithful_heal"] + + +def _edit_manifest(project: str, content_hash: str, edit) -> None: + path = entry_dir(project, content_hash) / "manifest.json" + doc = json.loads(path.read_text()) + edit(doc) + path.write_text(json.dumps(doc, indent=2)) + + +def _evict(project: str, content_hash: str) -> Path: + """Delete the entry's snapshot (what the Cache page's delete does) and drop the in-process memo.""" + snap = baked_snapshot_path(project, content_hash) + assert snap is not None and snap.exists() + snap.unlink() + cached_result_expr.cache_clear() + return snap + + +def _install(project: str, big_source: Path, name: str = "big.parquet") -> None: + shutil.copy(big_source, data_dir(project) / name) + + +# --------------------------------------------------------------------------- +# recipes +# --------------------------------------------------------------------------- + + +def _agg_code(project: str) -> str: # Aggregate: worthy, deterministic + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.group_by("region").aggregate(total=t.price.sum(), n=t.count()) +""" + + +def _order_by_code(project: str) -> str: # a Sort over the 1.5M-row file: worthy, and more than one row group + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("big.parquet", project={project!r}) +expr = t.order_by("id") +""" + + +def _float_agg_code(project: str, *, grouped: bool) -> str: + aggregate = "t.group_by('g').aggregate" if grouped else "t.aggregate" + return f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("big.parquet", project={project!r}) +expr = {aggregate}(s=t.v.sum(), m=t.v.mean()) +""" + + +def _noisy_udf_code(project: str) -> str: + """A scalar UDF that returns a different value on every call: the entry is worthy and not reproducible.""" + return f""" +from tallyman_xorq.io import read_project_file +from xorq.expr.udf import make_pandas_udf +import xorq.vendor.ibis.expr.datatypes as dt +from xorq.vendor.ibis import schema as ibis_schema + +t = read_project_file("orders.parquet", project={project!r}) + + +def noisy(df): + import numpy as np + + return df["price"] * 0.0 + np.random.random(len(df)) + + +_udf = make_pandas_udf(noisy, ibis_schema({{"price": dt.float64}}), dt.float64, name="noisy") +expr = t.mutate(noise=_udf.on_expr(t)) +""" + + +def _counting_udf_code(project: str, counter: Path) -> str: + """A deterministic scalar UDF that appends one byte to ``counter`` per call, so a test can count executions. + + xorq pickles a UDF into the build, so a module-level counter would be a copy; a file is visible from outside. + """ + return f""" +from tallyman_xorq.io import read_project_file +from xorq.expr.udf import make_pandas_udf +import xorq.vendor.ibis.expr.datatypes as dt +from xorq.vendor.ibis import schema as ibis_schema + +COUNTER = {str(counter)!r} +t = read_project_file("orders.parquet", project={project!r}) + + +def bump(df): + with open(COUNTER, "ab") as fh: + fh.write(b"x") + return df["qty"] + 1 + + +_udf = make_pandas_udf(bump, ibis_schema({{"qty": dt.int64}}), dt.int64, name="bump") +expr = t.mutate(qty_plus=_udf.on_expr(t)) +""" + + +def _timestamp_code(project: str) -> str: + """An expression whose type is ``timestamp[s]``, which parquet stores as ``timestamp[ms]``.""" + return f""" +from tallyman_xorq.io import read_project_file +import xorq.vendor.ibis.expr.datatypes as dt + +t = read_project_file("events.parquet", project={project!r}) +expr = t.mutate(ts0=t.ts.cast(dt.Timestamp(scale=0))).order_by("id") +""" + + +# --------------------------------------------------------------------------- +# fixtures: one big source for the module, one built entry for the layout assertions +# --------------------------------------------------------------------------- + + +@pytest.fixture(scope="module") +def big_source(tmp_path_factory) -> Path: + return write_big_parquet(tmp_path_factory.mktemp("big_source") / "big.parquet") + + +@dataclass(frozen=True) +class _Built: + home: Path + project: str + content_hash: str + + +@pytest.fixture(scope="module") +def _big_built(tmp_path_factory, big_source): + """Build ``order_by("id")`` over the 1.5M-row source once, in a home of its own, for every layout test.""" + home = tmp_path_factory.mktemp("format_home") + patch = pytest.MonkeyPatch() + patch.setenv("TALLYMAN_HOME", str(home)) + try: + ensure_project("fmt") + _install("fmt", big_source) + result = build_and_persist("fmt", _order_by_code("fmt")) + yield _Built(home, "fmt", result.content_hash) + finally: + patch.undo() + + +@pytest.fixture +def big_built(_big_built, monkeypatch) -> _Built: + monkeypatch.setenv("TALLYMAN_HOME", str(_big_built.home)) + return _big_built + + +def _snapshot(built: _Built) -> Path: + snap = baked_snapshot_path(built.project, built.content_hash) + assert snap is not None and snap.exists() + return snap + + +def _row_group_sizes(path: Path) -> list[int]: + md = pq.ParquetFile(path).metadata + return [md.row_group(i).num_rows for i in range(md.num_row_groups)] + + +# --------------------------------------------------------------------------- +# D1: materialization is single-partition, so a float aggregate heals to the digest it was built with +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize("grouped", [False, True], ids=["ungrouped SUM and AVG", "group_by g SUM and AVG"]) +def test_float_aggregate_heals_to_the_digest_it_was_built_with(project, big_source, grouped): + """ADR-009 D1 (materialization runs single-partition): three heals, three matching digests, no false alarm. + + DataFusion sums each partition separately and merges the partial sums in arrival order, and float addition is not + associative, so on the default 14-partition connection every heal of this entry produced different low bits: a + different file, an ``unfaithful_heal`` record and a stat-cache wipe each time. Both shapes are checked because the + ungrouped accumulator and the grouped one merge differently. + """ + _install(project, big_source) + result = build_and_persist(project, _float_agg_code(project, grouped=grouped)) + h = result.content_hash + recorded = read_manifest(entry_dir(project, h)).result_digest + assert recorded, "a worthy entry records the digest of its snapshot" + + for attempt in range(3): + snap = _evict(project, h) + cached_result_expr(project, h) # the heal + assert snap.exists(), f"heal {attempt} did not write the snapshot back" + assert snapshot_file_digest(snap) == recorded, f"heal {attempt} produced a different digest than the build" + assert _unfaithful(project) == [] + assert recorded.startswith(PREFIX) # ADR-009 D2: the recorded value is a content digest + + +# --------------------------------------------------------------------------- +# D1: the plan runs on a single-partition connection that is not the process default +# --------------------------------------------------------------------------- + + +def test_single_partition_backend_is_one_partition_with_a_pinned_batch_size(): + """ADR-009 D1: ``SET datafusion.execution.target_partitions = 1`` and an explicit ``batch_size``. + + ``batch_size`` decides the batch boundaries an ungrouped float total sees (#187), so it is part of the + reproducibility contract along with the row-group size. + """ + materialize_module = _materialize_module() + assert materialize_module.SNAPSHOT_ROW_GROUP_ROWS == SNAPSHOT_ROW_GROUP + backend = materialize_module.single_partition_backend() + assert _show(backend, "datafusion.execution.target_partitions") == "1" + assert _show(backend, "datafusion.execution.batch_size") == str(materialize_module.SNAPSHOT_BATCH_SIZE) + assert materialize_module.single_partition_backend() is not backend, "each call is a fresh connection" + if (os.cpu_count() or 1) > 1: + assert _show(default_backend(), "datafusion.execution.target_partitions") != "1" + + +def test_materialize_runs_the_plan_on_the_single_partition_backend(project, orders_parquet, monkeypatch): + """ADR-009 D1: the loaded build is rebound onto the single-partition connection, not left on its own backends. + + ``load_expr`` mints its own backend objects, so a loaded build ignores a single-partition connection it was never + bound to. The spy records every connection ``materialize`` asks for. xorq registers a read's table on the backend + it executes on, so a connection with tables afterwards is one the plan ran on. The process default backend must + be left as it was. + """ + materialize_module = _materialize_module() + h = build_and_persist(project, _agg_code(project)).content_hash + before = _show(default_backend(), "datafusion.execution.target_partitions") + + seen = [] + real = materialize_module.single_partition_backend + + def spy(): + backend = real() + seen.append(backend) + return backend + + monkeypatch.setattr(materialize_module, "single_partition_backend", spy) + materialize_module.materialize(project, h) + + assert seen, "materialize never asked for a single-partition backend" + assert all(_show(b, "datafusion.execution.target_partitions") == "1" for b in seen) + assert any(b.list_tables() for b in seen), "the plan did not run on the single-partition backend" + assert _show(default_backend(), "datafusion.execution.target_partitions") == before + + +# --------------------------------------------------------------------------- +# D3: the snapshot's format +# --------------------------------------------------------------------------- + + +def test_snapshot_row_groups_are_1048576_rows_with_a_short_last_group(big_built): + """ADR-009 D3 (the snapshot's format): rows are regrouped into 1,048,576-row groups, not one group per batch. + + xorq's writer makes one Snappy row group per DataFusion batch (8,192 rows): 184 groups for this file, and a 3.68 GB + snapshot in the corpus has 9,525 groups and a 44.9 MB footer that every page request opens. + """ + sizes = _row_group_sizes(_snapshot(big_built)) + assert sizes[:-1] == [SNAPSHOT_ROW_GROUP] * (len(sizes) - 1), (len(sizes), sizes[:3]) + assert 0 < sizes[-1] <= SNAPSHOT_ROW_GROUP + assert sum(sizes) == BIG_ROWS + assert len(sizes) == 2 + + +def test_snapshot_is_zstd_format_2_6_with_statistics_and_a_page_index(big_built): + """ADR-009 D3 and ADR-008 D5: zstd level 3, parquet format 2.6, statistics on, and a parquet page index. + + The page index is what takes a range request on ``__row_order`` from about 90 ms to about 20 ms. + """ + md = pq.ParquetFile(_snapshot(big_built)).metadata + assert md.format_version == "2.6" + for rg in range(md.num_row_groups): + for col in range(md.num_columns): + chunk = md.row_group(rg).column(col) + where = f"row group {rg}, column {chunk.path_in_schema}" + assert chunk.compression == "ZSTD", where + assert chunk.statistics is not None and chunk.statistics.has_min_max, where + assert chunk.has_offset_index and chunk.has_column_index, where + + +def test_snapshot_footer_is_small(big_built): + """ADR-009 D3: 2 row groups and 4 columns is a footer of about a kilobyte, not 60 KB.""" + assert pq.ParquetFile(_snapshot(big_built)).metadata.serialized_size < 8_192 + + +def test_snapshot_ends_in_row_order_numbered_from_zero_in_file_order(big_built): + """ADR-009 D3 and ADR-008 D2: the writer's last column is ``__row_order``, int64, ``0..N-1`` in the file's order.""" + table = pq.read_table(_snapshot(big_built)) + assert table.schema.names == ["id", "g", "v", "__row_order"] + assert table.schema.field("__row_order").type == pa.int64() + n = table.num_rows + assert n == BIG_ROWS + assert (table["__row_order"].to_numpy() == np.arange(n)).all() + # the recipe is order_by("id") over ids that are the source's positions, so file order and id order coincide + assert (table["id"].to_numpy() == np.arange(n)).all() + + +def test_ordered_copy_of_a_source_has_row_groups_of_122880_rows(big_built): + """ADR-009 D3 (the format version covers the ordered copies): polars writes them, in pinned 122,880-row groups. + + An ungrouped float total over a source reads that layout (#187), so it is as pinned as the snapshot's. + """ + copies = sorted((compute_cache_dir(big_built.project) / "ordered_sources").glob("*.parquet")) + assert copies, "reading a source should have written an ordered copy under compute_cache/ordered_sources/" + for copy in copies: + sizes = _row_group_sizes(copy) + assert sizes[:-1] == [ORDERED_COPY_ROW_GROUP] * (len(sizes) - 1), (copy.name, len(sizes), sizes[:3]) + assert 0 < sizes[-1] <= ORDERED_COPY_ROW_GROUP + assert pq.read_schema(copy).names[-1] == "__row_order" + + +def test_manifest_records_the_snapshot_format_version(project, orders_parquet): + """ADR-009 D3: row-group size and ``batch_size`` are contract; the manifest records a version for both.""" + materialize_module = _materialize_module() + result = build_and_persist(project, _agg_code(project)) + manifest = read_manifest(entry_dir(project, result.content_hash)) + assert manifest.snapshot_format == materialize_module.SNAPSHOT_FORMAT_VERSION + + +def test_materialize_returns_the_content_digest_of_the_file_it_wrote(project, orders_parquet): + """ADR-009 D2 and ADR-007 D4 (one writer): the digest is computed by one function, from the file read back.""" + materialize_module = _materialize_module() + h = build_and_persist(project, _agg_code(project)).content_hash + manifest = read_manifest(entry_dir(project, h)) + + out = materialize_module.materialize(project, h) + + assert out.path == baked_snapshot_path(project, h) == materialize_module.snapshot_path(project, h) + assert out.digest == snapshot_file_digest(out.path) + assert out.digest == manifest.result_digest + assert out.row_count == manifest.row_count + assert out.schema.names[-1] == "__row_order" + + +# --------------------------------------------------------------------------- +# D3: the recorded schema is the written file's, and every schema ends in __row_order +# --------------------------------------------------------------------------- + + +def test_recorded_schema_is_read_from_the_written_file(project): + """ADR-009 D3: parquet changes some types on the way in, and the file is what every consumer reads. + + The expression's type is ``timestamp[s]``; parquet has no seconds unit, so the file holds ``timestamp[ms]``. + """ + pq.write_table( + pa.table( + { + "id": np.arange(100), + "ts": pa.array(np.arange(100).astype("datetime64[s]")), + "v": np.random.default_rng(1).random(100), + } + ), + data_dir(project) / "events.parquet", + ) + h = build_and_persist(project, _timestamp_code(project)).content_hash + + recorded = {f["name"]: f["type"] for f in json.loads(entry_schema_path(project, h).read_text())["fields"]} + snap = baked_snapshot_path(project, h) + assert snap is not None + on_disk = {f.name: str(f.type) for f in pq.read_schema(snap)} + + assert on_disk["ts0"] == "timestamp[ms]" + assert recorded == on_disk + + +@pytest.mark.parametrize( + ("body", "expected"), + [ + ("expr = t.mutate(total=t.price * t.qty)", ["order_id", "region", "category", "price", "qty", "total"]), + ("expr = t.group_by('region').aggregate(n=t.count())", ["region", "n"]), + ("expr = t.order_by('price')", ["order_id", "region", "category", "price", "qty"]), + ("expr = t.mutate(rn=ibis.row_number())", ["order_id", "region", "category", "price", "qty", "rn"]), + ], + ids=["cheap computed column", "aggregate", "sort", "window function"], +) +def test_every_recorded_schema_ends_in_row_order(project, orders_parquet, body, expected): + """ADR-009 D3 and ADR-008 D2: cheap entries carry ``__row_order`` last, worthy ones get it from the writer. + + A worthy entry that keeps its parent's rows inherits ``__row_order`` mid-table and the writer replaces it with a + fresh last column; a cheap entry's computed column must not push it into the middle (ADR-008 D3). + """ + code = f""" +import xorq.vendor.ibis as ibis +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +{body} +""" + h = build_and_persist(project, code).content_hash + names = [f["name"] for f in json.loads(entry_schema_path(project, h).read_text())["fields"]] + assert names == [*expected, "__row_order"] + + +# --------------------------------------------------------------------------- +# D4: engine versions, and an unfaithful heal that says so +# --------------------------------------------------------------------------- + + +def test_manifest_records_the_engine_versions(project, orders_parquet): + """ADR-009 D4 (a mismatch record names its likely cause): xorq, xorq-datafusion and pyarrow at build.""" + result = build_and_persist(project, _agg_code(project)) + recorded = read_manifest(entry_dir(project, result.content_hash)).engine_versions + assert recorded is not None + for key, distribution in (("xorq", "xorq"), ("xorq_datafusion", "xorq-datafusion"), ("pyarrow", "pyarrow")): + assert recorded[key] == version(distribution), key + + +@pytest.mark.parametrize("engine_changed", [True, False], ids=["engine version changed", "engine versions match"]) +def test_an_unfaithful_heal_names_an_engine_change_instead_of_blaming_the_recipe( + project, orders_parquet, engine_changed +): + """ADR-009 D4: with an upgrade in play the record says so, and the loud response of ADR-006 D7 still happens. + + The recorded digest is replaced by one no heal can match, so the heal is unfaithful by construction. When the + manifest's recorded xorq version is not the installed one, the message names the engine, not the recipe's + "execution" or "structural" nondeterminism, which today it blames. + """ + h = build_and_persist(project, _agg_code(project)).content_hash + recorded_versions = dict(read_manifest(entry_dir(project, h)).engine_versions) + + def tamper(doc: dict) -> None: + doc["result_digest"] = PREFIX + "0" * 64 + if engine_changed: + doc["engine_versions"] = {**recorded_versions, "xorq": "0.0.1"} + + _edit_manifest(project, h, tamper) + fired = [] + + def hook(hook_project: str, hook_hash: str) -> None: + fired.append((hook_project, hook_hash)) + + UNFAITHFUL_HEAL_HOOKS.append(hook) + try: + _evict(project, h) + cached_result_expr(project, h) + finally: + UNFAITHFUL_HEAL_HOOKS.remove(hook) + + records = _unfaithful(project) + assert len(records) == 1 and records[0]["hash"] == h, records + assert fired == [(project, h)] + message = records[0]["message"] + if engine_changed: + assert "xorq" in message.lower(), message + assert "0.0.1" in message or version("xorq") in message, message + assert "execution (#83)" not in message and "structural (#88)" not in message, message + + +# --------------------------------------------------------------------------- +# D6: create runs the query twice and compares +# --------------------------------------------------------------------------- + + +def test_create_records_a_non_reproducible_recipe_and_names_the_column(project, orders_parquet): + """ADR-009 D6 (create runs the query twice and compares): the build succeeds, the entry says it is not reproducible. + + A recipe that calls ``sample()`` is legitimate, so the build still succeeds. The columns whose per-column digests + differ between the two runs are named, and the file is kept. + """ + result = build_and_persist(project, _noisy_udf_code(project)) + assert result.reproducible is False + assert result.nonreproducible_columns == ["noise"] + + manifest = read_manifest(entry_dir(project, result.content_hash)) + assert manifest.reproducible is False + assert manifest.nonreproducible_columns == ["noise"] + assert manifest.result_digest and manifest.result_digest.startswith(PREFIX) + snap = baked_snapshot_path(project, result.content_hash) + assert snap is not None and snap.exists() + + +def test_create_records_a_deterministic_recipe_as_reproducible(project, orders_parquet): + """ADR-009 D6: a recipe that runs the same both times is recorded as reproducible, with no offending columns.""" + result = build_and_persist(project, _agg_code(project)) + assert result.reproducible is True + assert result.nonreproducible_columns == [] + manifest = read_manifest(entry_dir(project, result.content_hash)) + assert manifest.reproducible is True + assert not manifest.nonreproducible_columns + + +def test_a_cheap_entry_is_not_checked_for_reproducibility(project, orders_parquet): + """ADR-009 D6: a cheap entry is not run twice here; it records no digest either (ADR-006 D9, no cheap digests).""" + code = f""" +from tallyman_xorq.io import read_project_file +t = read_project_file("orders.parquet", project={project!r}) +expr = t.mutate(total=t.price * t.qty) +""" + result = build_and_persist(project, code) + assert result.reproducible is None + manifest = read_manifest(entry_dir(project, result.content_hash)) + assert manifest.reproducible is None + assert manifest.result_digest is None + + +def test_create_runs_the_query_twice_and_a_heal_runs_it_once(project, orders_parquet, tmp_path): + """ADR-009 D6: only a create runs the query twice, since a create has nothing recorded to compare against. + + A heal is compared against the recorded digest, so once is enough. The UDF appends a byte to a file on every call. + """ + counter = tmp_path / "calls.bin" + counter.write_bytes(b"") + h = build_and_persist(project, _counting_udf_code(project, counter)).content_hash + at_create = len(counter.read_bytes()) + + counter.write_bytes(b"") + snap = _evict(project, h) + cached_result_expr(project, h) + at_heal = len(counter.read_bytes()) + + assert snap.exists() + assert at_heal > 0 + assert at_create == 2 * at_heal, f"create called the UDF {at_create} times, a heal {at_heal}" + + +def test_materialize_runs_once_and_runs_twice_only_when_asked_to_check(project, orders_parquet, tmp_path): + """ADR-009 D6 and ADR-007 D4: the checking run is an option of the one writer, not a second code path.""" + materialize_module = _materialize_module() + counter = tmp_path / "calls.bin" + counter.write_bytes(b"") + h = build_and_persist(project, _counting_udf_code(project, counter)).content_hash + + counter.write_bytes(b"") + materialize_module.materialize(project, h) + once = len(counter.read_bytes()) + assert once > 0 + + counter.write_bytes(b"") + checked = materialize_module.materialize(project, h, check_reproducible=True) + assert len(counter.read_bytes()) == 2 * once + assert checked.reproducible is True + assert checked.differing_columns == [] + + +def test_materialize_names_the_columns_that_differ_between_the_two_runs(project, orders_parquet): + """ADR-009 D6: ``differing_columns`` is what the build result reports to the author.""" + materialize_module = _materialize_module() + h = build_and_persist(project, _noisy_udf_code(project)).content_hash + checked = materialize_module.materialize(project, h, check_reproducible=True) + assert checked.reproducible is False + assert checked.differing_columns == ["noise"] + + +# --------------------------------------------------------------------------- +# verify: the digest a heal is checked against is the content digest +# --------------------------------------------------------------------------- + + +def test_verify_result_faithful_follows_the_content_not_the_bytes(project, orders_parquet): + """ADR-009 D2 in the verify path: the same rows in another format are faithful, a changed value is not. + + ``verify_result_faithful`` runs on every heal and in the corpus sweep. Under a byte hash, rewriting the snapshot + with a different codec and row-group size reads as drift; under the content digest it does not. + """ + h = build_and_persist(project, _agg_code(project)).content_hash + manifest = read_manifest(entry_dir(project, h)) + assert manifest.result_digest.startswith(PREFIX) + assert verify_result_faithful(project, h) is True + + snap = baked_snapshot_path(project, h) + assert snap is not None + assert snapshot_file_digest(snap) == manifest.result_digest + table = pq.read_table(snap) + + pq.write_table(table, snap, compression="snappy", row_group_size=2) + assert verify_result_faithful(project, h) is True + + changed = table.set_column(table.schema.get_field_index("n"), "n", pc.add(table["n"], 1)) + pq.write_table(changed, snap) + assert verify_result_faithful(project, h) is False + + +def test_verify_result_faithful_is_false_after_the_recorded_digest_is_tampered(project, orders_parquet): + """ADR-009 D2: the recorded value is an ``arrow-sha256:`` digest, and a wrong one fails verification.""" + h = build_and_persist(project, _agg_code(project)).content_hash + assert read_manifest(entry_dir(project, h)).result_digest.startswith(PREFIX) + assert verify_result_faithful(project, h) is True + + _edit_manifest(project, h, lambda doc: doc.update(result_digest=PREFIX + "0" * 64)) + assert verify_result_faithful(project, h) is False + + +# --------------------------------------------------------------------------- +# D6: a pinned file survives an explicit delete +# --------------------------------------------------------------------------- + + +def test_a_non_reproducible_entrys_snapshot_survives_an_explicit_delete(fresh_companion_app, project, orders_parquet): + """ADR-009 D6: the Cache page's delete skips a file that cannot be recreated, and says why. + + A snapshot whose recipe is not reproducible would come back as different rows, and everything built on the + original would then disagree with it. The listing flags the row as pinned. + """ + h = build_and_persist(project, _noisy_udf_code(project)).content_hash + snap = baked_snapshot_path(project, h) + assert snap is not None and snap.exists() + client = TestClient(fresh_companion_app) + + response = client.delete(f"/{project}/api/result_cache/{h}") + + assert response.status_code == 409, response.text + assert "reproducible" in response.json()["detail"].lower() + assert snap.exists() + rows = {row["hash"]: row for row in client.get(f"/{project}/api/result_cache").json()["entries"]} + assert rows[h]["pinned"] is True and rows[h]["pinned_reason"] + + +def test_an_entry_whose_heal_was_unfaithful_is_pinned_against_an_explicit_delete( + fresh_companion_app, project, orders_parquet +): + """ADR-009 D6 and ADR-006 D12 (unfaithful entries are pinned and badged): the same pin, reached after the fact.""" + h = build_and_persist(project, _agg_code(project)).content_hash + _edit_manifest(project, h, lambda doc: doc.update(result_digest=PREFIX + "0" * 64)) + _evict(project, h) + cached_result_expr(project, h) # an unfaithful heal: the durable record is the pin + assert len(_unfaithful(project)) == 1 + snap = baked_snapshot_path(project, h) + assert snap is not None and snap.exists() + + response = TestClient(fresh_companion_app).delete(f"/{project}/api/result_cache/{h}") + + assert response.status_code == 409, response.text + assert snap.exists() diff --git a/tests/test_tallyman_read_csv.py b/tests/test_tallyman_read_csv.py index b2f7053..53c2518 100644 --- a/tests/test_tallyman_read_csv.py +++ b/tests/test_tallyman_read_csv.py @@ -1,7 +1,10 @@ -"""Tests for tallyman_read_csv — CSV ingest with canonical row ordering. +"""Tests for tallyman_read_csv — CSV ingest with a stable ``__row_order``. -ADR-004-result-digest-canonical-ordering: tallyman_read_csv injects -``original_row_order`` so snapshot bytes are deterministic across builds. +ADR-004-result-digest-canonical-ordering gave a CSV read an ``original_row_order`` column and a trailing ``order_by`` +so that a snapshot's bytes are deterministic. ADR-008 (plans/ADR-008-row-order-of-reads.md) replaces both: the column +is ``__row_order`` (ADR-008 D7, amending ADR-005 INV-1), the trailing sort is gone so a CSV root is a cheap read +(ADR-008 D7, amending ADR-005 INV-2), and the ordered copy of the CSV lives in the project's +``compute_cache/ordered_sources`` (ADR-007 D13), keyed by the CSV's content (ADR-008 D2, #168). """ from __future__ import annotations @@ -9,9 +12,17 @@ import pytest -from tallyman_core import data_dir, entry_dir +from tallyman_core import data_dir, entry_dir, read_manifest +from tallyman_core.paths import compute_cache_dir, tallyman_home from tallyman_mcp.server import catalog_create from tallyman_xorq.build import list_entries +from tallyman_xorq.result_cache import ( + baked_snapshot_path, + cache_worthy, + cached_result_expr, + snapshot_file_digest, + verify_result_faithful, +) @pytest.fixture @@ -19,8 +30,8 @@ def sample_csv(project: str) -> Path: """Small CSV under the project data dir with a known row ordering.""" p = data_dir(project) / "sample.csv" # Deliberately write rows in an order that is NOT alphabetical by name, so - # any test that verifies sorted-by-original_row_order can distinguish a - # correctly-ordered snapshot from an arbitrarily-ordered one. + # any test that verifies file order can distinguish a correctly-ordered + # copy from an arbitrarily-ordered one. p.write_text("id,name,value\n3,charlie,30\n1,alice,10\n2,bob,20\n") return p @@ -38,116 +49,88 @@ def _read_csv_code(csv_path: Path) -> str: """ -def test_tallyman_read_csv_adds_original_row_order(project, sample_csv, monkeypatch): - """tallyman_read_csv returns an expression whose schema includes original_row_order: int64.""" +def _ordered_copies(project: str) -> list[Path]: + """The ordered copies of sources: ADR-007 D13 puts them under the project's compute cache.""" + return sorted((compute_cache_dir(project) / "ordered_sources").glob("*.parquet")) + + +def test_tallyman_read_csv_adds_row_order(project, sample_csv, monkeypatch): + """tallyman_read_csv returns an expression whose schema ends in __row_order: int64 (ADR-008 D7).""" monkeypatch.setenv("TALLYMAN_PROJECT", project) res = catalog_create("csv_entry", _read_csv_code(sample_csv)) assert "error" not in res, res + names = [f["name"] for f in res["schema"]["fields"]] schema_fields = {f["name"]: f["type"] for f in res["schema"]["fields"]} - assert "original_row_order" in schema_fields, ( - f"original_row_order missing from schema; got {list(schema_fields)}" - ) - assert schema_fields["original_row_order"] == "int64" + assert names == ["id", "name", "value", "__row_order"], f"one row-order column, and it is last: {names}" + assert schema_fields["__row_order"] == "int64" + assert "original_row_order" not in schema_fields -def test_tallyman_read_csv_entry_is_worthy(project, sample_csv, monkeypatch): - """An entry built with tallyman_read_csv is cache-worthy (RowNumber → Sort/Window op).""" - from tallyman_xorq.result_cache import cache_worthy +def test_tallyman_read_csv_entry_is_cheap(project, sample_csv, monkeypatch): + """A tallyman_read_csv entry is a plain read of its ordered copy, so it is cheap (ADR-008 D7). + Before ADR-008 its trailing ``order_by`` made every entry in a CSV lineage worthy for that Sort alone, and each + revision baked a full sorted copy. + """ monkeypatch.setenv("TALLYMAN_PROJECT", project) catalog_create("csv_entry", _read_csv_code(sample_csv)) h = _hash_of(project) - assert cache_worthy(project, h) is True, ( - "tallyman_read_csv entry must be cache-worthy (RowNumber adds a window op)" - ) - + assert cache_worthy(project, h) is False, "a tallyman_read_csv root has no Sort, so it is a cheap read" + assert read_manifest(entry_dir(project, h)).cache_worthy is False -def test_tallyman_read_csv_always_bakes_snapshot(project, sample_csv, monkeypatch): - """tallyman_read_csv entries ALWAYS bake a parquet snapshot (parquet reads beat CSV re-scans). - - The RowNumber + WindowFunction ops from ibis.row_number() are in _EXPENSIVE_OPS, so there is - no code path where a tallyman_read_csv entry ends up cheap with no snapshot. - """ - from tallyman_xorq.result_cache import baked_snapshot_path, cache_worthy +def test_tallyman_read_csv_bakes_no_snapshot(project, sample_csv, monkeypatch): + """A CSV root writes no snapshot: its rows are fixed by the content-keyed ordered copy (ADR-008 D7).""" monkeypatch.setenv("TALLYMAN_PROJECT", project) res = catalog_create("csv_entry", _read_csv_code(sample_csv)) assert "error" not in res, res h = _hash_of(project) - # Must be worthy — no path where a tallyman_read_csv entry is cheap. - assert cache_worthy(project, h) is True, "tallyman_read_csv entry must always be cache-worthy" - - # The snapshot must exist on disk immediately after the build (not lazy). - snap = baked_snapshot_path(project, h) - assert snap is not None, "baked_snapshot_path must return a path for a worthy entry" - assert snap.exists(), f"snapshot must be on disk after build; expected at {snap}" + assert baked_snapshot_path(project, h) is None, "a cheap entry has no snapshot" + snapshots = compute_cache_dir(project) / "result_cache" + assert not snapshots.exists() or not list(snapshots.glob("*.parquet")), "nothing was materialized" -def test_tallyman_read_csv_snapshot_is_sorted(project, sample_csv, monkeypatch): - """The baked snapshot rows are in ascending original_row_order order.""" - from tallyman_xorq.result_cache import baked_snapshot_path +def test_tallyman_read_csv_ordered_copy_is_in_file_order(project, sample_csv, monkeypatch): + """The ordered copy holds the CSV's rows in file order, numbered 0..N-1 in ``__row_order`` (ADR-008 D2).""" + import pyarrow.parquet as pq monkeypatch.setenv("TALLYMAN_PROJECT", project) catalog_create("csv_entry", _read_csv_code(sample_csv)) - h = _hash_of(project) - - snap = baked_snapshot_path(project, h) - assert snap is not None and snap.exists(), "worthy entry must bake a snapshot" - # Read the snapshot directly as a parquet to check physical row order. - import pyarrow.parquet as pq - - table = pq.read_table(str(snap)) - row_orders = table.column("original_row_order").to_pylist() - assert row_orders == sorted(row_orders), ( - f"snapshot rows not in ascending original_row_order; got {row_orders}" - ) - # Also verify the values are 0-based and contiguous. - assert row_orders == list(range(len(row_orders))), ( - f"expected 0-based contiguous row orders; got {row_orders}" - ) + copies = _ordered_copies(project) + assert len(copies) == 1, f"expected one ordered copy under compute_cache/ordered_sources, found {copies}" + table = pq.read_table(str(copies[0])) + assert table.column_names == ["id", "name", "value", "__row_order"] + assert table["id"].to_pylist() == [3, 1, 2], "file order, not sorted order" + assert table["__row_order"].to_pylist() == [0, 1, 2] -def test_tallyman_read_csv_digest_is_stable(project, sample_csv, monkeypatch): - """Building the same CSV entry twice yields the same result_digest. +def test_a_recreated_ordered_copy_matches_its_recorded_digest(project, sample_csv, monkeypatch): + """The manifest records the copy's content digest, and a re-created copy reproduces it (ADR-007 D13). - This tests byte-stability: both builds read the same source, inject the same - row numbers, sort by them, and bake — so the file hash must match. + Replaces the check that a CSV root's snapshot digest is stable across a heal: a CSV root is cheap now, so the + file whose reproducibility matters is the ordered copy. """ - from tallyman_core import read_manifest - from tallyman_xorq.result_cache import baked_snapshot_path, snapshot_file_digest - - monkeypatch.setenv("TALLYMAN_SOURCE_IDENTITY", "off") monkeypatch.setenv("TALLYMAN_PROJECT", project) - - res1 = catalog_create("csv_entry", _read_csv_code(sample_csv)) - assert "error" not in res1, res1 + res = catalog_create("csv_entry", _read_csv_code(sample_csv)) + assert "error" not in res, res h = _hash_of(project) - recorded_digest = read_manifest(entry_dir(project, h)).result_digest - assert recorded_digest, "worthy entry must record a result_digest" - - snap = baked_snapshot_path(project, h) - assert snap is not None and snap.exists() - - # The recorded digest must equal the file hash. - assert snapshot_file_digest(snap) == recorded_digest, ( - "recorded digest does not match snapshot file hash" - ) - - # Evict and self-heal to get a second materialisation. - from tallyman_xorq.result_cache import cached_result_expr + records = read_manifest(entry_dir(project, h)).ordered_copies + assert records and len(records) == 1, f"the manifest must record the ordered copy: {records}" + ((key, record),) = records.items() + recorded = record["content_digest"] + assert recorded.startswith("arrow-sha256:"), recorded - snap.unlink() + copy = compute_cache_dir(project) / "ordered_sources" / f"{key}.parquet" + assert copy.exists() and snapshot_file_digest(copy) == recorded + copy.unlink() cached_result_expr.cache_clear() - cached_result_expr(project, h).execute() # forces self-heal + cached_result_expr(project, h) # opening the entry makes the copy again from the clone - assert snap.exists(), "self-heal must rematerialise the snapshot" - healed_digest = snapshot_file_digest(snap) - assert healed_digest == recorded_digest, ( - f"digest changed between build and self-heal: {recorded_digest!r} vs {healed_digest!r}" - ) + assert copy.exists(), "opening the entry must re-create its ordered copy" + assert snapshot_file_digest(copy) == recorded, "the re-created copy must reproduce the recorded digest" @pytest.fixture @@ -155,7 +138,7 @@ def repartitioned_csv(project: str) -> tuple[Path, int]: """A CSV large enough that datafusion repartitions the scan across threads. The marker column holds the source row index (0..N-1) in file order, so a - correct ``original_row_order`` must line up with it row-for-row. The file is + correct ``__row_order`` must line up with it row-for-row. The file is sized well over datafusion's ~10 MB ``repartition_file_min_size`` default; above that threshold (and with >1 core) the parallel CSV scan emits rows in nondeterministic arrival order, which a bare ``ibis.row_number()`` would @@ -183,110 +166,90 @@ def _big_read_csv_code(csv_path: Path) -> str: def test_tallyman_read_csv_preserves_file_order_under_repartition(project, repartitioned_csv, monkeypatch): - """original_row_order must equal true file order even when the scan repartitions. + """__row_order must equal true file order even when the scan repartitions. Regression for the canonical-ordering bug: a bare ``ibis.row_number()`` over a repartitioned datafusion CSV scan numbers rows in nondeterministic arrival - order, so ``order_by("original_row_order")`` reproduces that arbitrary order - rather than file order. With the marker column = source row index, the baked - snapshot's row at ``original_row_order == k`` must carry ``marker == k``. + order. With the marker column = source row index, the ordered copy's row at + ``__row_order == k`` must carry ``marker == k`` (polars numbers the rows in + file order, ADR-008 D2). """ import pyarrow.parquet as pq - from tallyman_xorq.result_cache import baked_snapshot_path - monkeypatch.setenv("TALLYMAN_PROJECT", project) csv_path, n = repartitioned_csv res = catalog_create("big_csv", _big_read_csv_code(csv_path)) assert "error" not in res, res - h = _hash_of(project) - snap = baked_snapshot_path(project, h) - assert snap is not None and snap.exists(), "worthy entry must bake a snapshot" - - table = pq.read_table(str(snap)).sort_by("original_row_order") - oro = table.column("original_row_order").to_pylist() + copies = _ordered_copies(project) + assert len(copies) == 1, f"expected one ordered copy under compute_cache/ordered_sources, found {copies}" + table = pq.read_table(str(copies[0])) + assert table.column("__row_order").to_pylist() == list(range(n)), "__row_order must be 0..N-1, contiguous" marker = table.column("marker").to_pylist() - assert oro == list(range(n)), "original_row_order must be 0..N-1, contiguous" - # The crux: row k of the source file must land at original_row_order == k. + # The crux: row k of the source file must land at __row_order == k. assert marker == list(range(n)), ( - "original_row_order does not match true file order — the scan reshuffle " + "__row_order does not match true file order — the scan reshuffle " "leaked into the row index (first divergence at " f"{next((i for i, m in enumerate(marker) if m != i), None)})" ) -def test_tallyman_read_csv_digest_stable_under_repartition(project, repartitioned_csv, monkeypatch): - """The snapshot digest is byte-stable across two independent materialisations of a repartitioned read.""" - from tallyman_core import read_manifest - from tallyman_xorq.result_cache import baked_snapshot_path, cached_result_expr, snapshot_file_digest - - monkeypatch.setenv("TALLYMAN_SOURCE_IDENTITY", "off") +def test_ordered_copy_digest_stable_under_repartition(project, repartitioned_csv, monkeypatch): + """A re-created ordered copy of a repartitioned CSV has the digest recorded when it was first written.""" monkeypatch.setenv("TALLYMAN_PROJECT", project) csv_path, _ = repartitioned_csv res = catalog_create("big_csv", _big_read_csv_code(csv_path)) assert "error" not in res, res h = _hash_of(project) - recorded = read_manifest(entry_dir(project, h)).result_digest - assert recorded, "worthy entry must record a result_digest" + ((key, record),) = read_manifest(entry_dir(project, h)).ordered_copies.items() + recorded = record["content_digest"] - snap = baked_snapshot_path(project, h) - assert snap is not None and snap.exists() - snap.unlink() + copy = compute_cache_dir(project) / "ordered_sources" / f"{key}.parquet" + assert copy.exists() + copy.unlink() cached_result_expr.cache_clear() - cached_result_expr(project, h).execute() # self-heal → second materialisation + cached_result_expr(project, h) # re-created from the clone by the read - assert snap.exists() - assert snapshot_file_digest(snap) == recorded, ( - "snapshot digest drifted across materialisations of a repartitioned read" + assert copy.exists() + assert snapshot_file_digest(copy) == recorded, ( + "the ordered copy's digest drifted across two independent ingests of a repartitioned CSV" ) def test_tallyman_read_csv_reconstructs_after_source_deleted(project, sample_csv, monkeypatch): - """#6: once the ordered parquet is baked, reconstruction must not touch the CSV. + """#6: once the ordered copy is written, reading the entry must not touch the CSV. - The CSV is read exactly once — at ingest. After the snapshot is baked, deleting - (or moving) the source CSV must not break path reconstruction: baked_snapshot_path - and verify_result_faithful re-run the recipe (-> tallyman_read_csv), and that must - resolve the already-written intermediate instead of stat-ing the now-absent CSV. + The CSV is read exactly once — at ingest. After that, deleting (or moving) the source CSV must not break the + entry: it is a cheap read of the ordered copy, so there is no snapshot to resolve and no digest to verify. """ - from tallyman_xorq.result_cache import baked_snapshot_path, verify_result_faithful - - monkeypatch.setenv("TALLYMAN_SOURCE_IDENTITY", "off") monkeypatch.setenv("TALLYMAN_PROJECT", project) res = catalog_create("csv_entry", _read_csv_code(sample_csv)) assert "error" not in res, res h = _hash_of(project) - assert baked_snapshot_path(project, h) is not None + assert baked_snapshot_path(project, h) is None, "a CSV root is a cheap entry: nothing is baked" - # Delete the source CSV; the baked snapshot and the ordered intermediate remain. + # Delete the source CSV; the ordered copy remains. sample_csv.unlink() - # Reconstruction must still resolve the snapshot (pre-fix: FileNotFoundError on src.stat()). - snap = baked_snapshot_path(project, h) - assert snap is not None and snap.exists(), "snapshot must resolve without the source CSV" - assert verify_result_faithful(project, h) is True, "faithful check must work without the CSV" + cached_result_expr.cache_clear() + df = cached_result_expr(project, h).execute() + assert df["id"].tolist() == [3, 1, 2], "the entry must read without the source CSV" + assert verify_result_faithful(project, h) is None, "a cheap entry records no digest to verify" -def test_ordered_csv_intermediate_lives_outside_project_dir(project, sample_csv, monkeypatch): - """#9: the ordered-CSV intermediate is keyed on the CSV path, not the active project. +def test_ordered_csv_copy_lives_in_the_project_compute_cache(project, sample_csv, monkeypatch): + """The ordered copy is cache, so it lives under the project's compute_cache (ADR-007 D13). - Placing it under a project's artifacts dir (via resolve_project(None)) means a - reconstruction under a *different* active project looks in the wrong dir and - re-reads the CSV. It must live in a project-independent location. + It used to live under TALLYMAN_HOME/csv_ordered, outside every project: never collected, not packed, and + outside the project root, so a CSV entry's build was not portable. """ - from tallyman_core import artifacts_dir - from tallyman_core.paths import tallyman_home - monkeypatch.setenv("TALLYMAN_PROJECT", project) res = catalog_create("csv_entry", _read_csv_code(sample_csv)) assert "error" not in res, res - shared = list((tallyman_home() / "csv_ordered").glob("*.parquet")) - assert shared, "ordered-CSV intermediate must live under TALLYMAN_HOME/csv_ordered" - proj_csv_ordered = artifacts_dir(project) / "csv_ordered" - assert not proj_csv_ordered.exists(), "intermediate must NOT be under the project artifacts dir" + assert _ordered_copies(project), "the ordered copy must live under compute_cache/ordered_sources" + assert not list((tallyman_home() / "csv_ordered").glob("*.parquet")), "csv_ordered under TALLYMAN_HOME is retired" def test_tallyman_read_csv_forwards_reader_kwargs(project, monkeypatch): @@ -307,29 +270,24 @@ def test_tallyman_read_csv_forwards_reader_kwargs(project, monkeypatch): res = catalog_create("semi", code) assert "error" not in res, res fields = {f["name"] for f in res["schema"]["fields"]} - assert {"id", "name", "original_row_order"} <= fields, ( + assert {"id", "name", "__row_order"} <= fields, ( f"separator=';' not forwarded — columns did not split: {fields}" ) # --------------------------------------------------------------------------- # -# Name collision: the CSV already carries an 'original_row_order' column. -# tallyman would add one via with_row_index, which raises a polars DuplicateError -# (not caught by the infer ladder). Instead: reuse the column when it IS the -# canonical 0..N-1 file sequence, and raise a clear, actionable error when the -# name is a coincidence carrying non-canonical data. +# The reserved name is now the exact string '__row_order' (ADR-008 D6). A CSV that already +# has a column of that name (a file tallyman exported) has it overwritten, with no +# validation of its values: the ordered copy numbers the rows in file order (ADR-008 D2). +# 'original_row_order' is not special any more: it is ordinary data. # --------------------------------------------------------------------------- # -def test_existing_row_order_column_reused(project, monkeypatch): - """A CSV already carrying a valid 0..N-1 'original_row_order' column ingests - by reusing that column, not by crashing on the row-index name collision.""" - import pyarrow.parquet as pq - - from tallyman_xorq.result_cache import baked_snapshot_path - +def test_existing_row_order_column_is_overwritten(project, monkeypatch): + """A CSV that already carries a '__row_order' column ingests with it overwritten by 0..N-1 in file order, + whatever its values were — no error, no validation, and still exactly one row-order column (last).""" monkeypatch.setenv("TALLYMAN_PROJECT", project) p = data_dir(project) / "hasorder.csv" - # Rows deliberately NOT sorted by name; original_row_order = true 0..N-1 order. - p.write_text("original_row_order,name\n0,charlie\n1,alice\n2,bob\n") + # Rows deliberately NOT sorted by name; the incoming __row_order values are not the file sequence. + p.write_text("__row_order,name\n7,charlie\n3,alice\n5,bob\n") code = f""" from tallyman_xorq.io import tallyman_read_csv expr = tallyman_read_csv({str(p)!r}) @@ -337,26 +295,20 @@ def test_existing_row_order_column_reused(project, monkeypatch): res = catalog_create("hasorder", code) assert "error" not in res, res h = _hash_of(project) - snap = baked_snapshot_path(project, h) - assert snap is not None and snap.exists() - table = pq.read_table(str(snap)).sort_by("original_row_order") - assert table.column("original_row_order").to_pylist() == [0, 1, 2] - assert table.column("name").to_pylist() == ["charlie", "alice", "bob"] + df = cached_result_expr(project, h).execute() + assert list(df.columns) == ["name", "__row_order"] + assert df["__row_order"].tolist() == [0, 1, 2] + assert df["name"].tolist() == ["charlie", "alice", "bob"] def test_existing_row_order_column_with_explicit_schema_accepted(project, monkeypatch): - """A CSV carrying a valid 0..N-1 'original_row_order' column plus an explicit - schema for its DATA columns must ingest. The reserved column is tallyman's, not - the caller's to spec, so the totality check must not demand it be named (pre-fix - it raised 'schema is not total — it leaves column(s) [original_row_order]').""" - import pyarrow.parquet as pq - - from tallyman_xorq.result_cache import baked_snapshot_path - + """A CSV carrying a '__row_order' column plus an explicit schema for its DATA columns must ingest. + The reserved column is tallyman's, not the caller's to spec, so the totality check must not demand it be + named.""" monkeypatch.setenv("TALLYMAN_PROJECT", project) p = data_dir(project) / "orderschema.csv" - # original_row_order present; rows deliberately NOT sorted by name. - p.write_text("original_row_order,id,name\n0,1,charlie\n1,2,alice\n2,3,bob\n") + # __row_order present; rows deliberately NOT sorted by name. + p.write_text("__row_order,id,name\n7,1,charlie\n3,2,alice\n5,3,bob\n") code = f""" from tallyman_xorq.io import tallyman_read_csv expr = tallyman_read_csv({str(p)!r}, schema={{"id": "int64", "name": "string"}}) @@ -364,74 +316,62 @@ def test_existing_row_order_column_with_explicit_schema_accepted(project, monkey res = catalog_create("orderschema", code) assert "error" not in res, res h = _hash_of(project) - snap = baked_snapshot_path(project, h) - assert snap is not None and snap.exists() - table = pq.read_table(str(snap)).sort_by("original_row_order") - assert table.column("original_row_order").to_pylist() == [0, 1, 2] - assert table.column("id").to_pylist() == [1, 2, 3] - assert table.column("name").to_pylist() == ["charlie", "alice", "bob"] - - -def test_existing_row_order_column_nonsequential_raises(project, monkeypatch): - """An 'original_row_order' column that is NOT the contiguous 0..N-1 sequence - (here 1..N) is a coincidental name clash — ingest must raise rather than - silently trust a non-canonical order.""" - monkeypatch.setenv("TALLYMAN_PROJECT", project) - p = data_dir(project) / "badorder.csv" - p.write_text("original_row_order,name\n1,alice\n2,bob\n3,charlie\n") - code = f""" -from tallyman_xorq.io import tallyman_read_csv -expr = tallyman_read_csv({str(p)!r}) -""" - res = catalog_create("badorder", code) - assert "error" in res - err = res["error"] - assert "original_row_order" in err - assert "reserved" in err.lower() or "0.." in err or "contiguous" in err.lower() + df = cached_result_expr(project, h).execute() + assert df["__row_order"].tolist() == [0, 1, 2] + assert df["id"].tolist() == [1, 2, 3] + assert df["name"].tolist() == ["charlie", "alice", "bob"] + +@pytest.mark.parametrize( + ("body", "values"), + [ + pytest.param("1,alice\n2,bob\n3,charlie\n", [1, 2, 3], id="not the 0..N-1 sequence"), + pytest.param("a,alice\nb,bob\n", ["a", "b"], id="not integers"), + ], +) +def test_original_row_order_is_ordinary_data(project, monkeypatch, body, values): + """'original_row_order' is not reserved any more: ADR-008 D6 reserves only the exact name '__row_order'. -def test_existing_row_order_column_noninteger_raises(project, monkeypatch): - """An 'original_row_order' column whose values aren't integers cannot be the - canonical row index — ingest must raise with the reserved-name message.""" + Before ADR-008 a CSV with such a column raised unless it was the canonical 0..N-1 sequence. + """ monkeypatch.setenv("TALLYMAN_PROJECT", project) - p = data_dir(project) / "strorder.csv" - p.write_text("original_row_order,name\na,alice\nb,bob\n") + p = data_dir(project) / "oro.csv" + p.write_text("original_row_order,name\n" + body) code = f""" from tallyman_xorq.io import tallyman_read_csv expr = tallyman_read_csv({str(p)!r}) """ - res = catalog_create("strorder", code) - assert "error" in res - err = res["error"] - assert "original_row_order" in err - assert "reserved" in err.lower() or "integer" in err.lower() or "0.." in err + res = catalog_create("oro", code) + assert "error" not in res, res + df = cached_result_expr(project, _hash_of(project)).execute() + assert df["original_row_order"].tolist() == values, "an ordinary column keeps its values" + assert df["__row_order"].tolist() == list(range(len(values))) @pytest.mark.parametrize( "schema", [ - (("original_row_order", "int64"), ("&rest", "infer")), # positional rename target - (("a", "int64"), ("original_row_order", "string")), # positional, second column + (("__row_order", "int64"), ("&rest", "infer")), # positional rename target + (("a", "int64"), ("__row_order", "int64")), # positional, second column ], ) -def test_schema_output_name_original_row_order_raises(project, monkeypatch, schema): - """A schema that maps a DATA column onto the reserved 'original_row_order' output - name collides with tallyman's auto-added row index. Pre-fix this leaked a raw - polars DuplicateError ('column original_row_order is duplicate'); ingest must - instead reject the reserved output name with a clear ValueError.""" +def test_schema_output_name_row_order_raises(project, monkeypatch, schema): + """A schema that maps a DATA column onto the reserved '__row_order' output name collides with tallyman's + row index: assigning to the column is not allowed (ADR-008 D6). Ingest must reject the reserved output + name with a clear ValueError.""" monkeypatch.setenv("TALLYMAN_PROJECT", project) from tallyman_xorq.io import tallyman_read_csv p = data_dir(project) / "renameoro.csv" - p.write_text("a,b\nx,10\ny,20\n") - with pytest.raises(ValueError, match="reserved"): + p.write_text("a,b\n1,10\n2,20\n") + with pytest.raises(ValueError, match="__row_order"): tallyman_read_csv(str(p), schema=schema) # --------------------------------------------------------------------------- # # Reserved-column exclusion, threaded consistently (review follow-ups). # -# The totality check excludes tallyman's reserved 'original_row_order' column so +# The totality check excludes tallyman's reserved '__row_order' column so # a complete DATA-column schema is not spuriously "not total". Three surfaces # read the header and must apply that same exclusion consistently, or the # recovery / diagnostic / positional-binding contracts break in exactly the path @@ -442,76 +382,72 @@ def test_schema_output_name_original_row_order_raises(project, monkeypatch, sche # --------------------------------------------------------------------------- # def test_suggested_schema_recovery_is_pasteable_with_reserved_column(project, monkeypatch): """#143 recovery contract, reserved-column path: when an explicit schema fails - to parse a CSV that carries a canonical 'original_row_order' column, the + to parse a CSV that carries a '__row_order' column, the suggested schema in the error must be paste-ready. The whole-file suggestion must NOT emit a cell for the reserved column (which the caller cannot spec) — - pre-fix it did, so pasting the suggestion back raised 'positional schema has 3 - columns but the CSV header has 2'.""" + pasting the suggestion back would otherwise over-count the columns.""" import ast monkeypatch.setenv("TALLYMAN_PROJECT", project) from tallyman_xorq.io import tallyman_read_csv p = data_dir(project) / "suggest_oro.csv" - # Canonical 0..N-1 original_row_order (trailing, as tallyman exports it); the - # amount column holds a float that an int64 pin cannot parse, so the - # explicit-mode suggestion path fires. - p.write_text("id,amount,original_row_order\n1,12.5,0\n2,3.0,1\n") + # __row_order present (as tallyman exports it); the amount column holds a float + # that an int64 pin cannot parse, so the explicit-mode suggestion path fires. + p.write_text("id,amount,__row_order\n1,12.5,0\n2,3.0,1\n") with pytest.raises(ValueError) as exc: tallyman_read_csv(str(p), schema=(("id", "int64"), ("amount", "int64"))) msg = str(exc.value) assert "Suggested schema" in msg, msg - assert "original_row_order" not in msg, f"suggestion leaked the reserved column: {msg}" + assert "__row_order" not in msg, f"suggestion leaked the reserved column: {msg}" - # The suggestion must paste back and parse (pre-fix it raised on the column count). + # The suggestion must paste back and parse. suggested = ast.literal_eval(msg.rsplit("schema=", 1)[-1].strip()) expr = tallyman_read_csv(str(p), schema=suggested) out = {k: str(v) for k, v in expr.schema().items()} assert out.get("amount") == "float64", out - assert "original_row_order" in out # the reserved column is still reused + assert "__row_order" in out # the reserved column is still there, numbered by tallyman @pytest.mark.parametrize( "schema", [ - (("a", "int64"), ("b", "int64"), ("c", "int64")), # over-long positional spec - {"zzz": "int64", "&rest": "infer"}, # by-name miss + pytest.param((("a", "int64"), ("b", "int64"), ("c", "int64")), id="over-long positional spec"), + # A guard: this one passes today, because the header listing in a by-name miss already names every column. + pytest.param({"zzz": "int64", "&rest": "infer"}, id="by-name miss (a guard)"), ], ) def test_schema_error_diagnostic_names_reserved_column(project, monkeypatch, schema): - """A schema error against a CSV that carries 'original_row_order' must not hide - that column. Pre-fix the diagnostic printed the reserved-stripped header, so a - 3-column file (a,b,original_row_order) was reported as 'has 2 [a, b]' — an - off-by-one that the user, staring at a 3-column file, cannot reconcile. The - reserved column must appear in the message.""" + """A schema error against a CSV that carries '__row_order' must not hide + that column. The diagnostic must not print the reserved-stripped header: a + 3-column file (a,b,__row_order) reported as 'has 2 [a, b]' is an off-by-one that the user, staring at a + 3-column file, cannot reconcile. The reserved column must appear in the message.""" monkeypatch.setenv("TALLYMAN_PROJECT", project) from tallyman_xorq.io import tallyman_read_csv p = data_dir(project) / "diag_oro.csv" - p.write_text("a,b,original_row_order\n1,2,0\n3,4,1\n") + p.write_text("a,b,__row_order\n1,2,0\n3,4,1\n") with pytest.raises(ValueError) as exc: tallyman_read_csv(str(p), schema=schema) - assert "original_row_order" in str(exc.value), ( + assert "__row_order" in str(exc.value), ( f"diagnostic hides the reserved column: {exc.value}" ) def test_positional_schema_rejects_nontrailing_reserved_column(project, monkeypatch): - """Positional cells bind by physical column position. A valid canonical - 'original_row_order' column that is NOT the last column shifts that mapping — - excluding it from the middle silently rebinds later cells onto the wrong data + """Positional cells bind by physical column position. A '__row_order' column that is NOT the last column + shifts that mapping — excluding it from the middle silently rebinds later cells onto the wrong data column (renaming/dropping a column with no error). Ingest must reject a positional schema in this layout rather than silently corrupt the output.""" monkeypatch.setenv("TALLYMAN_PROJECT", project) from tallyman_xorq.io import tallyman_read_csv p = data_dir(project) / "oro_middle.csv" - # original_row_order is canonical 0..N-1 (so _validate_existing_row_order - # passes) but sits in the MIDDLE, not trailing. - p.write_text("sku,original_row_order,qty\nA,0,10\nB,1,20\n") + # __row_order sits in the MIDDLE, not trailing. + p.write_text("sku,__row_order,qty\nA,0,10\nB,1,20\n") with pytest.raises(ValueError) as exc: tallyman_read_csv(str(p), schema=(("sku", "string"), ("row", "int64"))) msg = str(exc.value).lower() - assert "original_row_order" in msg and ("position" in msg or "last" in msg or "by-name" in msg), ( + assert "__row_order" in msg and ("position" in msg or "last" in msg or "by-name" in msg), ( f"expected a positional/last-column guard message; got: {exc.value}" ) From 5105def50b6c0542048365fa4776aa785cd653e6 Mon Sep 17 00:00:00 2001 From: Paddy Mullen Date: Mon, 21 Sep 2026 00:33:21 -0400 Subject: [PATCH 006/111] feat(core): re-entrant project lock, manifest fields, reset leaves compute_cache alone ADR-007 D11, D13, D14 and the manifest fields ADR-008 and ADR-009 record: - project_lock is public and re-entrant per thread, so a promote can build and then checkpoint, and a build can materialize, while another thread or process still waits. - Manifest loses snapshot_key and gains reproducible, nonreproducible_columns, snapshot_format, engine_versions and ordered_copies. - reset_to no longer records, prunes or restores compute_cache/ (no compute_cache.jsonl). Source clones no surviving entry refers to move to the bullpen instead of being deleted, and a reset forward copies them back. - ensure_cas_path clones to a unique temp name, so two builders cloning one source do not share one. - paths: entry_view_build_dir for the view build a worthy entry's grid is handed; buckaroo_sessions_path is gone (tallyman keeps no session record). Co-Authored-By: Claude Sonnet 5 --- src/tallyman_core/__init__.py | 4 + src/tallyman_core/catalog.py | 4 +- src/tallyman_core/catalog_state.py | 230 ++++++++++++++------------- src/tallyman_core/manifest.py | 46 +++--- src/tallyman_core/paths.py | 22 ++- src/tallyman_xorq/source_identity.py | 41 +++-- 6 files changed, 188 insertions(+), 159 deletions(-) diff --git a/src/tallyman_core/__init__.py b/src/tallyman_core/__init__.py index 71d7327..ce22a05 100644 --- a/src/tallyman_core/__init__.py +++ b/src/tallyman_core/__init__.py @@ -35,6 +35,7 @@ ENTRY_MANIFEST_FILENAME, ENTRY_SCHEMA_FILENAME, ENTRY_STAT_CACHE_DIRNAME, + ENTRY_VIEW_BUILD_DIRNAME, artifacts_dir, catalog_dir, data_dir, @@ -46,6 +47,7 @@ entry_manifest_path, entry_schema_path, entry_stat_cache_dir, + entry_view_build_dir, errors_path, exports_dir, list_projects, @@ -116,6 +118,7 @@ "entry_schema_path", "entry_stat_cache_dir", "entry_expanded_build_dir", + "entry_view_build_dir", "carry_forward_entry_config", "ENTRY_ARTIFACT_NAMES", "ENTRY_CACHE_NAMES", @@ -124,6 +127,7 @@ "ENTRY_SCHEMA_FILENAME", "ENTRY_STAT_CACHE_DIRNAME", "ENTRY_EXPANDED_BUILD_DIRNAME", + "ENTRY_VIEW_BUILD_DIRNAME", "errors_path", "exports_dir", "get_alias", diff --git a/src/tallyman_core/catalog.py b/src/tallyman_core/catalog.py index 03ad659..e27469d 100644 --- a/src/tallyman_core/catalog.py +++ b/src/tallyman_core/catalog.py @@ -55,7 +55,6 @@ "config.json", "notebook.jsonl", "entries.jsonl", - "compute_cache.jsonl", "prompts/*.jsonl", "post_processing/*.py", "post_processing/_disabled/*.py", @@ -68,8 +67,7 @@ # The repo .gitignore: keep the heavy/derived artifacts out of ``git add -A``. # Trailing-slash patterns match directories only, so ``entries/*/`` ignores the # build dirs while ``entries/.zip`` files stay trackable, and -# ``compute_cache/`` ignores the cache dir while ``compute_cache.jsonl`` (the -# pointer file) stays trackable. +# ``compute_cache/`` ignores the cache dir. GITIGNORE_LINES = ( "entries/*/", "bullpen/", diff --git a/src/tallyman_core/catalog_state.py b/src/tallyman_core/catalog_state.py index 9336b4d..475d7ac 100644 --- a/src/tallyman_core/catalog_state.py +++ b/src/tallyman_core/catalog_state.py @@ -8,20 +8,26 @@ ``git reset --hard`` restores them all at once. There is no longer a ``catalog.yaml`` round-trip (no ``capture``/``materialize`` of those sections). -What remains here is the *pointer* bookkeeping for the two **untracked** -artifact sets — the entry build dirs and the compute-cache warm set — in two -tracked JSONL files: +What remains here is the *pointer* bookkeeping for the untracked entry build +dirs, in one tracked JSONL file: entries.jsonl: {"hash": content_hash} # untracked entries// dirs - compute_cache.jsonl: {"path": relpath} # untracked compute-cache warm set Those heavy artifacts are content-addressed, additive, and gitignored, so ``git reset`` can't roll them back; ``reset_to`` reconciles them to the recorded -pointers via the bullpen — evictions retire (not deleted), and anything a +pointers via the bullpen: evictions retire (not deleted), and anything a restored step records but is missing comes back by copy. Live operations never -read the bullpen, so a re-added expression still computes cold. +read the bullpen. -A checkpoint captures the pointer lists, zips any pending recipe (catalog.py), +``compute_cache/`` is not managed here (ADR-007 D14). Its files are named by +content hash and each one can be made again by ``ensure_materialized``, so a +reset leaves them alone and a snapshot that is missing afterwards is healed and +verified like any other. The source clones under ``data/.cas/`` are data, the +only frozen copy of the bytes an entry was built from, so a reset moves the ones +no surviving entry refers to into the bullpen (never deletes them) and a reset +forward copies them back. + +A checkpoint captures the pointer list, zips any pending recipe (catalog.py), ``git add -A``, and commits once — through the fork-safe git primitive (git_util) under a per-project lock — keeping the four invariants #33 kept re-breaking structural rather than reviewed. @@ -36,6 +42,7 @@ import os import re import shutil +import threading from pathlib import Path from tallyman_core import catalog @@ -44,7 +51,7 @@ ENTRIES_DIRNAME, bullpen_dir, catalog_dir, - compute_cache_dir, + data_dir, entries_dir, ) @@ -60,7 +67,7 @@ # --------------------------------------------------------------------------- -# pointer bookkeeping: the two tracked JSONL files that replace catalog.yaml +# pointer bookkeeping: the tracked JSONL file that replaces catalog.yaml # --------------------------------------------------------------------------- @@ -68,10 +75,6 @@ def _entries_file(project: str) -> Path: return catalog_dir(project) / "entries.jsonl" -def _compute_cache_file(project: str) -> Path: - return catalog_dir(project) / "compute_cache.jsonl" - - def _read_jsonl(path: Path, field: str) -> list[str] | None: """The *field* of each line, or None when the file is absent — so callers can tell "never recorded" (no-op) from a recorded-empty list (reconcile to @@ -87,46 +90,32 @@ def _write_jsonl(path: Path, field: str, values: list[str]) -> None: def read_tallyman_state(project: str) -> dict: - """The pointer lists for the *untracked* artifacts the bullpen reconciles — - entry build dirs (``entry_hashes``) and the compute-cache warm set - (``compute_cache``). Each defaults to [] so callers never KeyError. + """The pointer list for the *untracked* entry build dirs the bullpen + reconciles (``entry_hashes``). Defaults to [] so callers never KeyError. The decomposed mutable sections (charts, display, post-processing, stats, aliases, notebook) are tracked files now, restored by ``git reset`` directly, so they are no longer carried here. """ - return { - "entry_hashes": _read_jsonl(_entries_file(project), "hash") or [], - "compute_cache": _read_jsonl(_compute_cache_file(project), "path") or [], - } + return {"entry_hashes": _read_jsonl(_entries_file(project), "hash") or []} -def write_tallyman_state( - project: str, *, entry_hashes: list[str] | None = None, compute_cache: list[str] | None = None -) -> None: - """Persist whichever pointer list is given to its tracked JSONL (the other - is left untouched).""" +def write_tallyman_state(project: str, *, entry_hashes: list[str] | None = None) -> None: + """Persist the entry pointer list to its tracked JSONL.""" if entry_hashes is not None: _write_jsonl(_entries_file(project), "hash", entry_hashes) - if compute_cache is not None: - _write_jsonl(_compute_cache_file(project), "path", compute_cache) # --------------------------------------------------------------------------- -# capture: live untracked-artifact listings -> the two tracked pointer files +# capture: the live entry listing -> the tracked pointer file # --------------------------------------------------------------------------- -def _list_cache_files(root: Path) -> list[str]: - if not root.exists(): - return [] - return sorted(str(p.relative_to(root)) for p in root.rglob("*") if p.is_file()) - - def capture_tallyman_state(project: str) -> dict: - """Snapshot the pointer lists for the untracked artifacts into their tracked - JSONL files. The decomposed sections write their own tracked files, so - capture no longer touches charts/display/pp/stats/aliases/notebook.""" + """Snapshot the entry pointer list into its tracked JSONL file. The + decomposed sections write their own tracked files, so capture no longer + touches charts/display/pp/stats/aliases/notebook, and it no longer lists + ``compute_cache/`` (ADR-007 D14), whose cost grew with the cache (#22).""" ed = entries_dir(project) # Only COMPLETE entry dirs (a manifest is the build's last write) — the same # filter zip_pending_entries uses, so capture and the zip writer agree on the @@ -136,13 +125,12 @@ def capture_tallyman_state(project: str) -> dict: entry_hashes = ( sorted(c.name for c in ed.iterdir() if c.is_dir() and (c / "manifest.json").is_file()) if ed.exists() else [] ) - compute_cache = _list_cache_files(compute_cache_dir(project)) - write_tallyman_state(project, entry_hashes=entry_hashes, compute_cache=compute_cache) - return {"entry_hashes": entry_hashes, "compute_cache": compute_cache} + write_tallyman_state(project, entry_hashes=entry_hashes) + return {"entry_hashes": entry_hashes} # --------------------------------------------------------------------------- -# prune/restore: reconcile untracked artifacts to the pointer lists, via the +# prune/restore: reconcile untracked artifacts to the pointer list, via the # bullpen — evictions are retired (moved), not destroyed, and a forward reset # copies the step's recorded set back instead of recomputing it. # --------------------------------------------------------------------------- @@ -180,30 +168,39 @@ def prune_entries(project: str) -> int: return removed -def prune_compute_cache(project: str) -> int: - valid = _read_jsonl(_compute_cache_file(project), "path") - if valid is None: - return 0 - valid = set(valid) - root = compute_cache_dir(project) - if not root.exists(): - return 0 - removed = 0 - for p in root.rglob("*"): - if p.is_file() and (rel := str(p.relative_to(root))) not in valid: - _retire(p, bullpen_dir(project) / "compute_cache" / rel) - removed += 1 - return removed +def _cas_bullpen(project: str) -> Path: + return bullpen_dir(project) / "cas" + + +def _live_source_digests(project: str) -> set[str] | None: + """The union of every surviving entry's ``manifest.sources`` digests, or None when a manifest can't be read. + + An unreadable manifest means the sweep would run on partial information, and a clone wrongly retired is the only + frozen copy of somebody's bytes, so callers skip the sweep. + """ + from tallyman_core.manifest import read_manifest + from tallyman_core.paths import entry_dir + + live: set[str] = set() + for h in read_tallyman_state(project)["entry_hashes"]: + try: + sources = read_manifest(entry_dir(project, h)).sources + except Exception: + return None + if sources: + live.update(sources.values()) + return live def restore_from_bullpen(project: str) -> int: """Copy recorded-but-missing artifacts back from the bullpen. - The inverse of the prunes, for a reset that walks forward: anything the - restored pointer files name that is absent from the live tree comes back by - *copy*, so the bullpen keeps its set and the back/forward rehearsal loop can - repeat. Only ``reset_to`` calls this — live operations never see the - bullpen, which is what keeps a re-added expression honest (cold). + The inverse of the prunes, for a reset that walks forward: every entry dir the + restored pointer file names that is absent from the live tree comes back by + *copy*, and so does every source clone (``data/.cas/``) that a restored entry + refers to, so the bullpen keeps its set and the back/forward rehearsal loop + can repeat. Only ``reset_to`` calls this — live operations never see the + bullpen. """ bp = bullpen_dir(project) restored = 0 @@ -213,13 +210,15 @@ def restore_from_bullpen(project: str) -> int: if not live.exists() and parked.is_dir(): shutil.copytree(parked, live) restored += 1 - root = compute_cache_dir(project) - for rel in _read_jsonl(_compute_cache_file(project), "path") or []: - live, parked = root / rel, bp / "compute_cache" / rel - if not live.exists() and parked.is_file(): - live.parent.mkdir(parents=True, exist_ok=True) - shutil.copy2(parked, live) - restored += 1 + parked_clones = _cas_bullpen(project) + live_digests = _live_source_digests(project) + if live_digests and parked_clones.is_dir(): + cas = data_dir(project) / ".cas" + for parked in parked_clones.iterdir(): + if parked.is_file() and parked.stem in live_digests and not (cas / parked.name).exists(): + cas.mkdir(parents=True, exist_ok=True) + shutil.copy2(parked, cas / parked.name) + restored += 1 return restored @@ -228,21 +227,48 @@ def restore_from_bullpen(project: str) -> int: # --------------------------------------------------------------------------- +# The project locks this thread holds, and how deep. flock takes a lock per open file description, so a nested +# acquire on a fresh descriptor would block forever behind the outer one: re-entrancy has to be counted per thread. +_held = threading.local() + + @contextlib.contextmanager -def _project_lock(project: str): - """Cross-process file lock so the companion and MCP server can't race a - checkpoint (the git index.lock race that swallowed #33 writes).""" +def project_lock(project: str): + """One write at a time per project (ADR-007 D11): a build, a materialization, a promote, a recalc, a checkpoint. + + A cross-process file lock, so it holds between the two processes of a normal tallyman (the MCP server and the + companion, which both build), and re-entrant within a thread, since a promote builds and then checkpoints and a + build materializes. Another thread or process waits. It is blocking with no timeout (ADR-007 D11, #186), and it + cannot be held across an ``await``: the companion moves work between threads with ``run_in_threadpool``. + """ + depth = getattr(_held, "depth", None) + if depth is None: + depth = _held.depth = {} + if depth.get(project, 0) > 0: + depth[project] += 1 + try: + yield + finally: + depth[project] -= 1 + return cd = catalog_dir(project) cd.mkdir(parents=True, exist_ok=True) fd = os.open(str(cd / ".checkpoint.lock"), os.O_CREAT | os.O_RDWR, 0o644) try: fcntl.flock(fd, fcntl.LOCK_EX) - yield + depth[project] = 1 + try: + yield + finally: + depth.pop(project, None) finally: fcntl.flock(fd, fcntl.LOCK_UN) os.close(fd) +_project_lock = project_lock # the name the lock had before it became public + + def ensure_catalog_repo(project: str) -> bool: """Idempotently git-init the catalog repo. Returns True if it exists after. @@ -286,7 +312,7 @@ def checkpoint_catalog(project: str, message: str, *, step: int | None = None, l which keys rows by commit. """ cd = catalog_dir(project) - with _project_lock(project): + with project_lock(project): ensure_catalog_repo(project) capture_tallyman_state(project) # The checkpoint is the sole zip writer and sole git transaction: zip any @@ -328,62 +354,46 @@ def reset_to(project: str, ref: int | str) -> None: """Restore the catalog to a step/label: ``git reset --hard`` (which restores every tracked file — the recipe zips and all decomposed mutable state — at once), then reconcile the *untracked* artifacts to the recorded pointers: - evictions retire to the bullpen, recorded-but-missing files come back from - it. Finally re-validate the tracked recipe set against the pointers, so a - step whose two views disagree fails loudly rather than returning a masked - divergence (#52).""" + evicted entry dirs retire to the bullpen, recorded-but-missing ones come back + from it, and so do the source clones a restored entry refers to. Clones no + surviving entry refers to are moved to the bullpen too, never deleted. + ``compute_cache/`` is left alone (ADR-007 D14). Finally re-validate the + tracked recipe set against the pointers, so a step whose two views disagree + fails loudly rather than returning a masked divergence (#52).""" cd = catalog_dir(project) tag = f"step-{ref:03d}" if isinstance(ref, int) else str(ref) - with _project_lock(project): + with project_lock(project): commit = _resolve_tag(project, tag) rc, _, err = run_git(["reset", "--hard", commit], cwd=cd) if rc != 0: raise RuntimeError(f"catalog reset to {tag!r} failed: {err}") prune_entries(project) - prune_compute_cache(project) restore_from_bullpen(project) - _gc_cas_clones(project) + _retire_cas_clones(project) catalog.assert_catalog_consistent(project, set(read_tallyman_state(project)["entry_hashes"])) - # The prune above retires baked snapshots on disk. Clear the in-process - # result-plan memo that resolved their paths: #96's literal complaint is that - # the LRU outlives the prune, so drop it. For the dangling-read *symptom* this - # clear is belt-and-suspenders — cached_result_expr re-checks path.exists() - # and self-heals on every read (ee0a90a), so its own reads already survive a - # prune. The load-bearing protection against the symptom is the companion's - # _build_compare_expr clear (#80): that is the one LRU that bakes the resolved - # path into a serialized build with no per-call recheck, and it lives in the - # companion layer (see app._invalidate_reset_caches), not here. Blunt global - # clear is correct and cheap: entries are content-addressed, so the next read - # rebuilds an identical plan. Lazy import avoids a core->xorq import cycle. + # Clear the in-process result-plan memo: a reset changes which entries exist, and a memoised plan for a retired + # entry would outlive it. Entries are content-addressed, so the next read rebuilds an identical plan. Lazy import + # avoids a core->xorq import cycle. from tallyman_xorq.result_cache import cached_result_expr # noqa: PLC0415 cached_result_expr.cache_clear() -def _gc_cas_clones(project: str) -> int: - """Reclaim content-addressed source clones (``data/.cas``) no surviving entry - references, after a reset has pruned the entry set. +def _retire_cas_clones(project: str) -> int: + """Move the source clones (``data/.cas``) no surviving entry references into the bullpen (ADR-007 D14). - ``.cas`` lives under ``data/`` — outside the catalog git repo — so the - ``git reset`` above cannot roll it back; this is the explicit reclaim. - Liveness is the union of every surviving entry's ``manifest.sources`` - digests. Conservative and best-effort: if any entry's manifest can't be read - we skip the sweep rather than risk deleting a live clone on partial - information, and a failure here never aborts the reset. + ``.cas`` lives under ``data/`` — outside the catalog git repo — so the ``git reset`` above cannot roll it back. + A clone is data, the only frozen copy of the bytes an entry was built from once the live file is edited, so + nothing deletes one: a reset forward brings it back (``restore_from_bullpen``). Liveness is the union of every + surviving entry's ``manifest.sources`` digests. Conservative and best-effort: if any entry's manifest can't be + read we skip the sweep, and a failure here never aborts the reset. """ - from tallyman_core.manifest import read_manifest - from tallyman_core.paths import entry_dir from tallyman_xorq import source_identity # lazy: avoid a core->xorq import cycle - live_digests: set[str] = set() - for h in read_tallyman_state(project)["entry_hashes"]: - try: - sources = read_manifest(entry_dir(project, h)).sources - except Exception: - return 0 # unreadable manifest — don't GC on partial information - if sources: - live_digests.update(sources.values()) - return source_identity.gc_cas(project, live_digests) + live_digests = _live_source_digests(project) + if live_digests is None: + return 0 + return source_identity.gc_cas(project, live_digests, bullpen=_cas_bullpen(project)) def genesis(project: str) -> int | None: diff --git a/src/tallyman_core/manifest.py b/src/tallyman_core/manifest.py index dc99fdc..ffc5cbe 100644 --- a/src/tallyman_core/manifest.py +++ b/src/tallyman_core/manifest.py @@ -36,33 +36,35 @@ class Manifest(BaseModel): schema_path: str = ENTRY_SCHEMA_FILENAME row_count: int | None = None execute_seconds: float | None = None - # Cache-admission instrumentation (#87): the two verdicts recorded side by - # side so the structural-vs-measured cache decision (#30) is decidable from - # data. cache_worthy / cache_worthy_why are the structural classify_build - # verdict (today computed and thrown away). compile_seconds (the author DAG's - # expr->backend-plan step, the dominant per-view cost) and cache_bytes (the - # baked snapshot size, the value-per-byte denominator) are the measured side; - # with execute_seconds they give recompute_cost. cache_bytes is None for a - # cheap entry that bakes no snapshot. All absent on entries built before #87. + # Whether the entry is materialized (ADR-008 D4): ``cache_worthy`` and its ``cache_worthy_why`` are the verdict of + # ``worthiness.classify_expr``, computed once on the live expression when the entry is built and read from here + # ever after. ``compile_seconds`` (the author DAG's expr->backend-plan step) and ``cache_bytes`` (the snapshot's + # size, None for a cheap entry that has no file) are the measured side of the cache-admission record (#87). compile_seconds: float | None = None cache_worthy: bool | None = None cache_worthy_why: str | None = None cache_bytes: int | None = None - # Order-sensitive content digest of the executed result bytes (#83), recorded - # at build as a second identity axis: content_hash keys on the expression - # *graph*, result_digest on the executed *bytes*. A cold recompute whose digest - # differs is execution nondeterminism (sample()/now()/unordered limit/impure - # UDF) the structural hash can't see — caught when a self-healed snapshot is - # re-checked. Absent on entries built before #83. + # Content digest of the materialized snapshot (ADR-009 D2), ``arrow-sha256:``: a SHA-256 over the file's + # ordered Arrow data, read back, so it does not move with the row-group size, the codec or the writer's version. + # content_hash keys on the expression *graph*, result_digest on the executed *result*. A rewrite of the snapshot + # whose digest differs is execution nondeterminism the structural hash can't see (sample()/now()/an impure UDF) + # or an engine change, caught when a healed snapshot is checked. None for a cheap entry, which has no file. result_digest: str | None = None - # Filename of the baked result-cache snapshot (the xorq cache key + .parquet), - # recorded at build so the canonical read can assert its own derivation matches - # (ADR D8, plans/ADR-006-read-path-loads-builds.md). Build and read share one - # derivation route; this is the tripwire that turns any future divergence (an - # xorq tokenization change, a rewrite drift) into an immediate, attributable - # failure instead of a silent wrong-file read. None for a cheap entry (bakes - # no snapshot) and under salt identity mode. - snapshot_key: str | None = None + # Whether two runs of the query at create time gave the same digest (ADR-009 D6). False pins the snapshot, since + # it cannot be re-created faithfully, and ``nonreproducible_columns`` names the columns whose digests differed. + # None for a cheap entry, which is not run twice. + reproducible: bool | None = None + nonreproducible_columns: list[str] | None = None + # The version of the snapshot format (row-group size and materialization batch size, ADR-009 D3) this entry's + # snapshot and ordered copies were written with, and the engine versions at build (ADR-009 D4), so a mismatch at a + # heal can say the engine changed instead of blaming the recipe. + snapshot_format: int | None = None + engine_versions: dict[str, str] | None = None + # Ordered copies of sources this entry's plan reads (ADR-007 D13, ADR-008 D2), keyed by the copy's file stem: + # ``{"source": rel_or_abs_path, "digest": md5 of the source, "suffix": ".csv", "reader": {"kind": "parquet"|"csv", + # ...}, "content_digest": "arrow-sha256:..."}``. What ``ensure_materialized`` needs to make a deleted copy again + # from its clone, and the digest the new copy is checked against. + ordered_copies: dict[str, dict] | None = None # rel data path -> content md5, recorded when a source-identity mode is # active (tallyman_xorq.source_identity); absent under mode=off. sources: dict[str, str] | None = None diff --git a/src/tallyman_core/paths.py b/src/tallyman_core/paths.py index cd0c287..a0fd0e9 100644 --- a/src/tallyman_core/paths.py +++ b/src/tallyman_core/paths.py @@ -4,7 +4,6 @@ ~/.tallyman/ ├── active_project # one-line plain text; source of truth - ├── buckaroo_sessions.json # global Buckaroo session map └── projects// ├── artifacts/ # everything the system produces │ ├── catalog/ # the native catalog git repo @@ -16,7 +15,7 @@ │ │ ├── display_configs/.json │ │ ├── post_processing/.py , stats/.py │ │ ├── prompts/.jsonl - │ │ └── entries.jsonl , compute_cache.jsonl # untracked-artifact pointers + │ │ └── entries.jsonl # untracked-artifact pointers │ ├── exports/... # marimo .py, screenshots, CSVs │ └── errors.jsonl └── data/ # input parquets (fixtures or user) @@ -53,16 +52,6 @@ def active_project_file_path() -> Path: return tallyman_home() / "active_project" -def buckaroo_sessions_path() -> Path: - """Global Buckaroo session table (one file, all projects). - - Schema: ``{: {session_id, project, buckaroo_started_at}}``. - Migrates to a Buckaroo-side enumeration endpoint once - buckaroo-data/buckaroo#860 lands. - """ - return tallyman_home() / "buckaroo_sessions.json" - - # --------------------------------------------------------------------------- # Layout segment + per-entry artifact/cache names (single source of truth) # --------------------------------------------------------------------------- @@ -84,6 +73,9 @@ def buckaroo_sessions_path() -> Path: # first hit is honestly cold; production deletes them to recompute. ENTRY_STAT_CACHE_DIRNAME = ".buckaroo_stat_cache" ENTRY_EXPANDED_BUILD_DIRNAME = ".xorq_build_expanded" +# The "view build" a worthy entry's grid is handed: a build whose whole graph is one bare read of the entry's snapshot +# (ADR-007 D6). Derived from the snapshot's path, so it is regenerated on demand like the expanded build. +ENTRY_VIEW_BUILD_DIRNAME = ".xorq_view_build" # No per-entry result.parquet exists: an expensive entry's rows live in its baked # result cache (under the per-project compute_cache), a cheap entry recomputes on # read. The single materialised copy is the xorq .cache() snapshot — nothing @@ -104,6 +96,7 @@ def buckaroo_sessions_path() -> Path: ENTRY_CACHE_NAMES = ( ENTRY_STAT_CACHE_DIRNAME, ENTRY_EXPANDED_BUILD_DIRNAME, + ENTRY_VIEW_BUILD_DIRNAME, ) @@ -206,6 +199,11 @@ def entry_expanded_build_dir(project: str, content_hash: str) -> Path: return entry_dir(project, content_hash) / ENTRY_EXPANDED_BUILD_DIRNAME +def entry_view_build_dir(project: str, content_hash: str) -> Path: + """Stable per-entry dir holding the view build of the entry's snapshot (regenerated on demand).""" + return entry_dir(project, content_hash) / ENTRY_VIEW_BUILD_DIRNAME + + def compute_cache_dir(project: str) -> Path: """Per-project xorq compute cache, redirected off the global ``~/.cache/xorq``. diff --git a/src/tallyman_xorq/source_identity.py b/src/tallyman_xorq/source_identity.py index 7061ed1..b2d4f6e 100644 --- a/src/tallyman_xorq/source_identity.py +++ b/src/tallyman_xorq/source_identity.py @@ -41,6 +41,7 @@ import shutil import subprocess import sys +import uuid from pathlib import Path from tallyman_core import artifacts_dir, data_dir @@ -134,9 +135,13 @@ def ensure_cas_path(project: str, src: Path, digest: str) -> Path: cas_dir.mkdir(parents=True, exist_ok=True) dst = cas_dir / f"{digest}{src.suffix}" if not dst.exists(): - tmp = dst.with_suffix(dst.suffix + ".tmp") - _clone(src, tmp) - os.replace(tmp, dst) + # A unique temp name (not a fixed .tmp): two builders cloning one source at once must not share one. + tmp = dst.with_name(f"{dst.name}.{uuid.uuid4().hex}.tmp") + try: + _clone(src, tmp) + os.replace(tmp, dst) + finally: + tmp.unlink(missing_ok=True) return dst @@ -174,28 +179,40 @@ def recon_cas_path(project: str, live_src: Path, digest: str) -> Path: return live_src -def gc_cas(project: str, live_digests: set[str]) -> int: - """Delete ``.cas`` clones whose digest no live entry references. +def gc_cas(project: str, live_digests: set[str], *, bullpen: Path | None = None) -> int: + """Retire ``.cas`` clones whose digest no live entry references. ``live_digests`` is the union of every live entry's ``manifest.sources`` values — the md5 the clone is named by (````). Returns the - number of files removed; a no-op when the ``.cas`` dir is absent. ``.cas`` + number of files retired; a no-op when the ``.cas`` dir is absent. ``.cas`` lives under ``data/``, outside the catalog git repo, so a reset's - ``git reset`` never reclaims it — this is the explicit reclaim, called from + ``git reset`` never reclaims it — this is the explicit sweep, called from ``reset_to`` against the post-prune live entry set. + + A clone is data (ADR-007 D13), the only frozen copy of the bytes an entry was built from once the live source is + edited, so with ``bullpen`` given it is MOVED there instead of deleted, and a reset forward copies it back. + Without one it is unlinked, which nothing in tallyman does any more. """ cas_dir = data_dir(project) / ".cas" if not cas_dir.is_dir(): return 0 - removed = 0 + retired = 0 for f in cas_dir.iterdir(): if f.is_file() and f.stem not in live_digests: try: - f.unlink() - removed += 1 + if bullpen is None: + f.unlink() + else: + bullpen.mkdir(parents=True, exist_ok=True) + dest = bullpen / f.name + if dest.exists(): + f.unlink() # content-addressed: an existing copy is the same bytes + else: + shutil.move(str(f), str(dest)) + retired += 1 except OSError: - pass # best-effort reclaim; never fail a reset over GC - return removed + pass # best-effort sweep; never fail a reset over it + return retired # --------------------------------------------------------------------------- From 4f544ff601827e80242f78c47933c749a085e68f Mon Sep 17 00:00:00 2001 From: Paddy Mullen Date: Mon, 21 Sep 2026 00:33:30 -0400 Subject: [PATCH 007/111] feat(xorq): tallyman-owned materialization, row order and content digest ADR-007 D1 to D5 and D7, ADR-008 D2 to D4, D6, D7 and D10 to D12, ADR-009 D1 to D4 and D6. - No build holds a xorq cache node. A recipe that calls .cache() is a build error; rewrite_cache_dirs, classify_build, snapshot_key, entry_graph_expr and the source-read injection are gone. - materialize is the one writer of snapshots: a single-partition connection, a pinned layout (zstd, 1,048,576-row groups, page index), a last __row_order column, atomic replace under the project lock. A create runs the query twice; a recipe that is not reproducible is recorded and its file is pinned. - ensure_materialized makes every file an entry reads exist before anything runs: snapshots by recursion, ordered copies of sources from their clone with the reader options in the manifest, clones from the live source while the bytes match. - Every source enters through an ordered copy under compute_cache/, keyed by content digest and reader options; tallyman_read_csv no longer sorts and its column is __row_order. A raw parquet read is a build error. - Cheap or worthy is one allow-list test decided at build and recorded in the manifest. A cheap entry must keep __row_order, assignment to it is an error, every order_by gets the natural order as tie-break and a non-final order_by is hoisted, and a three-way join in one recipe gets an instruction. - result_digest is an arrow-sha256 content digest of the file read back. A heal mismatch is attributed to an engine change when the recorded versions differ, and stays loud and pinned. - Chaining a worthy parent is a bare read of its snapshot, so a child's identity follows its parent's. Co-Authored-By: Claude Sonnet 5 --- src/tallyman_xorq/build.py | 257 +++++++------ src/tallyman_xorq/digest.py | 193 ++++++++++ src/tallyman_xorq/io.py | 372 +++++++----------- src/tallyman_xorq/materialize.py | 306 +++++++++++++++ src/tallyman_xorq/ordered_copy.py | 322 ++++++++++++++++ src/tallyman_xorq/portable.py | 42 +-- src/tallyman_xorq/primary_key.py | 10 +- src/tallyman_xorq/result_cache.py | 601 +++++++++++------------------- src/tallyman_xorq/row_order.py | 287 ++++++++++++++ src/tallyman_xorq/source_cache.py | 279 ++++---------- src/tallyman_xorq/staleness.py | 24 +- src/tallyman_xorq/worthiness.py | 86 +++++ 12 files changed, 1783 insertions(+), 996 deletions(-) create mode 100644 src/tallyman_xorq/digest.py create mode 100644 src/tallyman_xorq/materialize.py create mode 100644 src/tallyman_xorq/ordered_copy.py create mode 100644 src/tallyman_xorq/row_order.py create mode 100644 src/tallyman_xorq/worthiness.py diff --git a/src/tallyman_xorq/build.py b/src/tallyman_xorq/build.py index 27a2751..4638d66 100644 --- a/src/tallyman_xorq/build.py +++ b/src/tallyman_xorq/build.py @@ -92,6 +92,10 @@ class BuildResult: cache_worthy: bool | None = None cache_worthy_why: str | None = None cache_bytes: int | None = None + # Whether two runs of the query at create time gave the same digest (ADR-009 D6): None for a cheap entry, which is + # not run twice, False when the recipe is not reproducible, with the columns whose digests differed. + reproducible: bool | None = None + nonreproducible_columns: list[str] = field(default_factory=list) class BuildError(RuntimeError): @@ -115,7 +119,7 @@ def _user_imports_bare_ibis(code: str) -> bool: # Read helpers the model reaches for on the wrong namespace. The project-aware -# way is read_project_file/tracked_expr_from_alias; xo.deferred_read_parquet handles a raw path. +# way is read_project_file/tracked_expr_from_alias/tallyman_read_csv. _READ_FNS = frozenset( {"read_parquet", "read_csv", "read_in_memory", "read_delta", "deferred_read_parquet", "deferred_read_csv"} ) @@ -196,14 +200,14 @@ def _ibis_import_hint(exc_msg: str, code: str = "") -> str: "(`import xorq.api as xo`). Read data with `from tallyman_xorq.io " "import read_project_file, tracked_expr_from_alias, tallyman_read_csv`; " "use `tallyman_read_csv(abs_path, schema=...)` for CSVs (the only " - "supported CSV ingest) and `xo.deferred_read_parquet(abs_path)` " - "for a raw parquet path." + "supported CSV ingest) and `read_project_file(rel_path)` for a " + "parquet file under /data/." ) else: hints.append( f"`xorq.{name}` does not exist — `xorq` is not the API entrypoint. " f"`import xorq.api as xo` and call `xo.{name}` " - "(e.g. xo.memtable, xo.deferred_read_parquet, xo.connect)." + "(e.g. xo.memtable, xo.connect)." ) m = re.search(r"module 'ibis' has no attribute '(\w+)'", exc_msg) @@ -231,7 +235,7 @@ def _ibis_import_hint(exc_msg: str, code: str = "") -> str: f"`tallyman_xorq.io` has no `{m.group(1)}` — it exports `read_project_file` " "(raw files under /data/), `tracked_expr_from_alias` (alias → records lineage), " "`pinned_expr_from_alias` (hash or 'name-vN' version ref, pinned), and `tallyman_read_csv` " - "(CSV ingest with original_row_order for stable digests)." + "(CSV ingest with a stable __row_order)." ) if "duckdb" in exc_msg.lower(): @@ -275,10 +279,9 @@ def _ibis_import_hint(exc_msg: str, code: str = "") -> str: def _csv_direct_read_check(expr) -> None: """Raise BuildError if the recipe calls xo.deferred_read_csv directly. - tallyman_read_csv is the only supported CSV ingest path — it bakes an - order-stable polars intermediate so result_digest is byte-reproducible - across builds. A raw deferred_read_csv bypasses that and produces a - nondeterministic row order above datafusion's repartition threshold. + tallyman_read_csv is the only supported CSV ingest path — it goes through source identity and writes an ordered + copy with a stable ``__row_order``, so the entry's rows are fixed under its hash and paging is repeatable. A raw + deferred_read_csv bypasses that and gives a nondeterministic row order above datafusion's repartition threshold. """ try: from xorq.expr.relations import Read @@ -294,12 +297,39 @@ def _csv_direct_read_check(expr) -> None: raise BuildError( "xo.deferred_read_csv is not allowed in tallyman recipes — " "use tallyman_read_csv(abs_path, schema=...) instead. " - "tallyman_read_csv bakes an order-stable polars intermediate so " - "result_digest is byte-reproducible; deferred_read_csv gives " + "tallyman_read_csv ingests the CSV once into an ordered copy with a stable __row_order, so the " + "entry's rows are fixed under its hash; deferred_read_csv gives " "nondeterministic row order above datafusion's repartition threshold." ) +def _raw_parquet_read_check(expr, project: str) -> None: + """Raise BuildError if the recipe reads a parquet file that tallyman did not write (ADR-008 D12). + + Such a read has no digest, no clone and no ordered copy, so an entry built on it has no ``__row_order`` to page by. + Tallyman's own files are the snapshots and ordered copies under the project's ``compute_cache/``; everything else + goes through ``read_project_file``. + """ + from xorq.common.utils.graph_utils import walk_nodes + from xorq.expr.relations import Read + + from tallyman_core.paths import compute_cache_dir + + root = compute_cache_dir(project).resolve() + for node in walk_nodes(Read, expr): + if node.method_name != "read_parquet": + continue + path = dict(node.read_kwargs).get("hash_path") + if path is None or Path(str(path)).resolve().is_relative_to(root): + continue + raise BuildError( + f"the recipe reads {path} with xo.deferred_read_parquet, which is not allowed: the file gets no " + "content digest, no clone and no ordered copy, so the entry would have no __row_order to page by. " + "Read a parquet file under /data/ with read_project_file('.parquet'), and a catalog " + "entry with tracked_expr_from_alias('')." + ) + + def _nondeterminism_warnings(expr) -> list[str]: """Advisory lint: flag execution-nondeterministic ops in a recipe (#88). @@ -364,68 +394,105 @@ def build_and_persist( expr_name: str = "expr", prompt: str | None = None, ) -> BuildResult: - """Compile user code with xorq, materialize a parquet, write a catalog entry. + """Compile user code with xorq, materialize a worthy entry's snapshot, write a catalog entry. The user code must bind a variable named `expr_name` (default "expr") to an ibis/xorq expression. Imports happen in a fresh module scope. + + The whole build holds the project's write lock (ADR-007 D11): one write at a time per project, so two builds of + one entry cannot end with the failing one deleting the winner's directory, and a chained build waits for the + materialization of its parent. """ + from tallyman_core.catalog_state import project_lock + + ensure_project(project) + with project_lock(project): + return _build_and_persist(project, code, expr_name, prompt) + + +def _reading(expr, project: str, ordered: dict[str, dict]) -> str: + """What a cheap entry reads, in words, for the message that says it must keep ``__row_order`` (ADR-008 D3).""" + from xorq.common.utils.graph_utils import walk_nodes + from xorq.expr.relations import Read + + from tallyman_xorq.ordered_copy import describe_read + + paths = [dict(r.read_kwargs).get("hash_path") for r in walk_nodes(Read, expr)] + paths = [Path(str(p)) for p in paths if p] + return describe_read(project, paths[0], ordered) if len(paths) == 1 else "a file" + + +def _build_and_persist(project: str, code: str, expr_name: str, prompt: str | None) -> BuildResult: from xorq.ibis_yaml.compiler import build_expr, load_expr - from tallyman_core.paths import compute_cache_dir + from tallyman_xorq import ordered_copy as oc from tallyman_xorq._git_state_guard import install_git_state_guard + from tallyman_xorq.materialize import SNAPSHOT_FORMAT_VERSION, engine_versions, materialize, snapshot_path + from tallyman_xorq.result_cache import stream_row_count + from tallyman_xorq.row_order import RowOrderError + from tallyman_xorq.worthiness import classify_expr # git-provenance capture in xorq's compiler can crash (git SIGSEGV when forked # from the long-lived server) and abort the whole build. Make it best-effort. install_git_state_guard() - ensure_project(project) - # Collect source digests while user code imports (read_project_file notes each - # file it reads); cas/salt identity modes consume them below. + # file it reads), and the ordered copies each source was ingested into (what a + # manifest needs so ensure_materialized can make one again). from tallyman_xorq import parent_capture as pc from tallyman_xorq import source_identity as si collect_token = si.begin_collect() parent_token = pc.begin_collect() + copies_token = oc.begin_collect() try: module, tmp_script = _import_script(code) finally: sources = si.end_collect(collect_token) # Resolved tracked_expr_from_alias parent edges captured during import (#84). parents = pc.end_collect(parent_token) + ordered = oc.end_collect(copies_token) expr_obj = getattr(module, expr_name, None) if expr_obj is None: names = ", ".join(n for n in dir(module) if not n.startswith("_")) raise BuildError(f"variable {expr_name!r} not found in code. Available names: {names}") - # Advisory nondeterminism lint (#88) on the author's expression, before the - # rewrite wraps it in cache nodes. Surfaced on the result, never fatal. + # Advisory nondeterminism lint (#88) on the author's expression, surfaced on the result, never fatal. lint_warnings = _nondeterminism_warnings(expr_obj) - # Fatal: raw deferred_read_csv is banned — tallyman_read_csv is the only - # supported CSV ingest path (order-stable polars intermediate). + # Fatal: raw reads are banned. tallyman_read_csv and read_project_file are the only ingest paths (each writes an + # ordered copy with a stable __row_order); a raw deferred_read_csv or deferred_read_parquet bypasses them. _csv_direct_read_check(expr_obj) + _raw_parquet_read_check(expr_obj, project) - # Rewrite-then-build (#73): cache each non-parquet source read, bake a - # top-level result cache when the expression is expensive (so every loader, - # incl. the Buckaroo viewer, reads the cached result instead of re-running - # the DAG), and reject in-memory reads. The content_hash below is computed - # from this rewritten expression; xorq_build/ carries the cache nodes, while - # expr.py keeps the author's literal source (tallyman does not depend on the - # submitted form). - from tallyman_xorq.source_cache import InMemoryReadError, rewrite_for_build - - # The author's DAG before cache injection — compile_seconds (#87) times its - # expr->backend-plan step, the un-truncated "large expression DAG" recompile - # that #30's profiling found dominates per-view cost. The rewritten expression - # carries cache boundaries that would truncate that compile. + # Whether the entry is materialized is decided ONCE, here, on the expression the author wrote (ADR-008 D4), and + # recorded in the manifest. The rewrite below then adds the canonical sort to a worthy entry and checks that a + # cheap one keeps __row_order; no cache node is created (ADR-007 D1). + from tallyman_xorq.source_cache import CacheNodeError, InMemoryReadError, rewrite_for_build + + verdict = classify_expr(expr_obj) + # The author's DAG before the canonical sort: compile_seconds (#87) times its expr->backend-plan step, the + # un-truncated "large expression DAG" recompile that #30's profiling found dominates per-view cost. author_expr = expr_obj try: - expr_obj = rewrite_for_build(expr_obj, project) - except InMemoryReadError as exc: + expr_obj = rewrite_for_build( + expr_obj, + project, + verdict=verdict, + reading=_reading(expr_obj, project, ordered) if not verdict.worthy else None, + ) + except (InMemoryReadError, CacheNodeError, RowOrderError) as exc: raise BuildError(str(exc)) from exc + except Exception as exc: + from tallyman_xorq.row_order import translate_collision + + translated = translate_collision(exc) + if translated is not None: + raise BuildError(str(translated)) from exc + raise created_target = False + wrote_snapshot: Path | None = None try: # Use a temp builds_dir so xorq's hash naming doesn't collide; we move # things into our catalog layout afterwards. @@ -462,6 +529,8 @@ def build_and_persist( cache_worthy=meta.get("cache_worthy"), cache_worthy_why=meta.get("cache_worthy_why"), cache_bytes=meta.get("cache_bytes"), + reproducible=meta.get("reproducible"), + nonreproducible_columns=meta.get("nonreproducible_columns") or [], ) target.mkdir(parents=True, exist_ok=True) @@ -486,90 +555,49 @@ def build_and_persist( code_persisted = code.replace(str(project_dir(project)), "${TALLYMAN_PROJECT_ROOT}") (target / "expr.py").write_text(code_persisted) - # Load + execute. - try: - # Per-project compute cache (not the global ~/.cache/xorq), so a - # reset can prune it to the revision's warm-set and a freshly added - # expression computes cold. See plans/adr-reset-to-revision.md. - cache_dir = compute_cache_dir(project) - cache_dir.mkdir(parents=True, exist_ok=True) - loaded = load_expr(build_path, cache_dir=cache_dir) - except Exception as exc: - hint = _ibis_import_hint(str(exc), code) - raise BuildError(f"load_expr failed: {exc}{hint}\n{traceback.format_exc()}") from exc - - # Classify before executing (reads the serialized build, no eval): the - # verdict selects the execution strategy below and is reused for the #87 - # admission instrumentation after the temp dir is gone. - from tallyman_xorq.result_cache import _cached_node_path, classify_build # noqa: PLC0415 - - verdict = classify_build(xorq_build_dir) - cache_worthy_v, cache_worthy_why = verdict["worthy"], verdict["why"] - - # Execute (#73). A worthy entry's build carries a baked result-cache - # node (rewrite_for_build), so executing materialises it once into the - # compute cache — and every later loader (the Buckaroo viewer, diffs, - # tracked_expr_from_alias) reads that cached result instead of re-running the DAG; - # baking evaluates every row, so count() there both validates and counts. - # A cheap entry materialises nothing, so count() alone is satisfied from - # source metadata and PRUNES the row projection — a failing cast / - # arithmetic / UDF would never evaluate at build, committing + aliasing a - # broken entry that throws only on the first materialising read. Stream - # the full result and discard it to force row-level evaluation at author - # time (constant memory; the one pass yields the exact row count). - # - # result_digest (ADR-004-result-digest-canonical-ordering): for worthy - # entries the baked snapshot is sorted by original_row_order before - # materialisation, so its bytes are deterministic run-to-run and we - # hash the file directly (cheap, ~0.3s). Cheap entries record no digest - # — their result has no snapshot to hash. - from tallyman_xorq.result_cache import snapshot_file_digest, stream_row_count # noqa: PLC0415 - + # Execute (ADR-007 D4). A worthy entry is materialized: the ONE writer runs the frozen build on a + # single-partition connection and writes the snapshot, twice, so a recipe that is not reproducible is known + # from birth (ADR-009 D6). A cheap entry writes nothing: one full streaming pass forces row-level evaluation + # at author time (a failing cast / arithmetic / UDF surfaces here, in tallyman's process, and not later in + # a grid query), and the one pass yields the exact row count. + cache_worthy_v, cache_worthy_why = verdict.worthy, verdict.why + reproducible: bool | None = None + differing: list[str] = [] t0 = time.monotonic() try: - arrow_schema = loaded.schema().to_pyarrow() if cache_worthy_v: - # count() bakes the snapshot (the cache node forces full - # materialisation); then hash the baked file for the digest. - # rewrite_for_build injected the canonical sort under the - # CachedNode (ADR D5, amended), so the bake lands in a - # deterministic total order and the file hash is stable - # across the build and every later heal. snapshot_key - # records the baked file's name so the canonical read can - # assert its own derivation matches (ADR D8). - row_count = int(loaded.count().execute()) - snap_path = _cached_node_path(loaded) - baked_ok = snap_path is not None and snap_path.exists() - result_digest_v: str | None = snapshot_file_digest(snap_path) if baked_ok else None - snapshot_key_v: str | None = snap_path.name if baked_ok else None + wrote_snapshot = snapshot_path(project, content_hash) + result = materialize(project, content_hash, check_reproducible=True) + row_count = result.row_count + result_digest_v: str | None = result.digest + arrow_schema = result.schema + reproducible, differing = result.reproducible, result.differing_columns else: - # Stream full result to force row-level evaluation (catch a - # failing cast / arithmetic at build time) and count rows. - # No digest for cheap entries — no snapshot to hash. + loaded = load_expr(build_path) + arrow_schema = loaded.schema().to_pyarrow() row_count = stream_row_count(loaded) result_digest_v = None - snapshot_key_v = None except Exception as exc: hint = _ibis_import_hint(str(exc), code) + from tallyman_xorq.row_order import translate_collision + + translated = translate_collision(exc) + if translated is not None: + raise BuildError(str(translated)) from exc raise BuildError(f"build execution failed: {exc}{hint}\n{traceback.format_exc()}") from exc execute_seconds = round(time.monotonic() - t0, 3) - # Schema + row count come from the expression directly (no result.parquet to - # read back); a worthy entry's rows are already materialised in its cache. + # Schema + row count: a worthy entry's schema is read from the file it wrote (parquet changes some types, and + # the writer adds __row_order), a cheap entry's from the expression, which by the D3 check already ends in it. schema_doc = { "fields": [{"name": f.name, "type": str(f.type)} for f in arrow_schema], "row_count": row_count, } atomic_write_text(entry_schema_path(project, content_hash), json.dumps(schema_doc, indent=2)) - # Cache-admission instrumentation (#87): record the structural verdict - # (computed above, before execute, since it now also selects the execution - # strategy) alongside the measured value-per-byte inputs, so #30's - # structural→measured flip is decidable from data. compile_seconds times the - # author DAG's expr->backend-plan step (the dominant per-view cost per #30), - # separate from execute; cache_bytes is the baked snapshot size (None for a - # cheap entry that bakes nothing). loaded is the build_path expression whose - # execute just baked the snapshot, so its CachedNode names that exact file. + # Cache-admission instrumentation (#87): record the structural verdict alongside the measured inputs. + # compile_seconds times the author DAG's expr->backend-plan step (the dominant per-view cost per #30), + # separate from execute; cache_bytes is the snapshot size (None for a cheap entry that writes nothing). compile_seconds: float | None = None try: from xorq.expr.api import to_sql @@ -582,10 +610,9 @@ def build_and_persist( pass cache_bytes: int | None = None - snap = _cached_node_path(loaded) - if snap is not None and snap.exists(): + if cache_worthy_v: try: - cache_bytes = snap.stat().st_size + cache_bytes = snapshot_path(project, content_hash).stat().st_size except OSError: pass @@ -610,7 +637,11 @@ def build_and_persist( cache_worthy_why=cache_worthy_why, cache_bytes=cache_bytes, result_digest=result_digest_v, - snapshot_key=snapshot_key_v, + reproducible=reproducible, + nonreproducible_columns=differing or None, + snapshot_format=SNAPSHOT_FORMAT_VERSION, + engine_versions=engine_versions(), + ordered_copies=ordered or None, sources=sources or None, parents=parents or None, ) @@ -624,9 +655,21 @@ def build_and_persist( # durable entry, which always early-returns before created_target is set. if created_target: shutil.rmtree(target, ignore_errors=True) + if wrote_snapshot is not None: + wrote_snapshot.unlink(missing_ok=True) raise _append_prompt(project, content_hash, prompt) + if reproducible is False: + lint_warnings = [ + *lint_warnings, + "this entry's query is not reproducible: two runs at create time gave different results in column(s) " + + ", ".join(differing) + + ". The entry was built, and its snapshot file is pinned (never deleted by the Cache page) because it " + "cannot be re-created faithfully. Seed a sample, or replace now()/random()/an impure UDF with a value " + "fixed at author time, for a reproducible entry.", + ] + # Best-effort: drop the temp script. try: tmp_script.unlink() @@ -652,6 +695,8 @@ def build_and_persist( cache_worthy=cache_worthy_v, cache_worthy_why=cache_worthy_why, cache_bytes=cache_bytes, + reproducible=reproducible, + nonreproducible_columns=differing, ) diff --git a/src/tallyman_xorq/digest.py b/src/tallyman_xorq/digest.py new file mode 100644 index 0000000..5226ca9 --- /dev/null +++ b/src/tallyman_xorq/digest.py @@ -0,0 +1,193 @@ +"""Content digest of a parquet file (ADR-009 D2): what ``result_digest`` is. + +A SHA-256 over the file's ordered Arrow data, read back. It does not depend on how the rows were batched or grouped +when the file was written, on the codec, on the writer's version, on whether a text column is ``string``, +``large_string`` or ``string_view``, or on what a null slot happens to hold. It does depend on every value, on which +slots are null, on the order of the rows and on the column names and logical types. + +Each column keeps one hash stream per kind of byte: validity (one byte per row), lengths (variable-width types) and +values. One hasher per column would interleave the three chunk by chunk, and the digest would then depend on where +the batch boundaries fell. A nested column keeps the same three streams for itself and one set for each child. + +The digest of a column and the digest of the file are both exposed: the file digest is what ``manifest.result_digest`` +records, and the per-column digests name the columns that differ between two runs of a recipe that is not +reproducible (ADR-009 D6). +""" + +from __future__ import annotations + +import hashlib +from pathlib import Path + +import numpy as np +import pyarrow as pa +import pyarrow.compute as pc +import pyarrow.parquet as pq + +ALGORITHM = "arrow-sha256" +_STREAMS = ("validity", "lengths", "values") +_READ_BATCH_ROWS = 65_536 + + +def logical_type(dtype: pa.DataType) -> str: + """The type a column is hashed under: physical spellings of one logical type collapse to one name.""" + if pa.types.is_dictionary(dtype): + return logical_type(dtype.value_type) + if pa.types.is_string(dtype) or pa.types.is_large_string(dtype) or pa.types.is_string_view(dtype): + return "string" + if pa.types.is_binary(dtype) or pa.types.is_large_binary(dtype) or pa.types.is_binary_view(dtype): + return "binary" + if pa.types.is_map(dtype): + return f"map<{logical_type(dtype.key_type)},{logical_type(dtype.item_type)}>" + if _is_list(dtype): + return f"list<{logical_type(dtype.value_type)}>" + if pa.types.is_struct(dtype): + return "struct<" + ",".join(f"{f.name}:{logical_type(f.type)}" for f in dtype) + ">" + return str(dtype) + + +def _is_list(dtype: pa.DataType) -> bool: + return ( + pa.types.is_list(dtype) + or pa.types.is_large_list(dtype) + or pa.types.is_fixed_size_list(dtype) + or pa.types.is_list_view(dtype) + or pa.types.is_large_list_view(dtype) + ) + + +class _ColumnHasher: + """The three streams of one column, plus the hashers of its children when the column is nested.""" + + def __init__(self, label: str, dtype: pa.DataType): + logical = logical_type(dtype) + self.streams = {s: hashlib.sha256(f"{label}\x00{logical}\x00{s}".encode()) for s in _STREAMS} + self.children: list[_ColumnHasher] = [] + self.dtype = dtype + if pa.types.is_dictionary(dtype): + self.dtype = dtype.value_type + if pa.types.is_map(self.dtype): + entries = pa.struct([pa.field("key", self.dtype.key_type), pa.field("value", self.dtype.item_type)]) + self.children = [_ColumnHasher(f"{label}.entries", entries)] + elif _is_list(self.dtype): + self.children = [_ColumnHasher(f"{label}.item", self.dtype.value_type)] + elif pa.types.is_struct(self.dtype): + self.children = [_ColumnHasher(f"{label}.{f.name}", f.type) for f in self.dtype] + + def update(self, arr: pa.Array | pa.ChunkedArray) -> None: + if isinstance(arr, pa.ChunkedArray): + for chunk in arr.chunks: + self.update(chunk) + return + if pa.types.is_dictionary(arr.type): + arr = arr.dictionary_decode() + nulls = arr.is_null().to_numpy(zero_copy_only=False) + self.streams["validity"].update(nulls.astype(np.uint8).tobytes()) + dtype = arr.type + if pa.types.is_null(dtype): + return + if pa.types.is_string(dtype) or pa.types.is_large_string(dtype) or pa.types.is_string_view(dtype): + self._update_varlen(arr.cast(pa.large_string()), b"") + elif pa.types.is_binary(dtype) or pa.types.is_large_binary(dtype) or pa.types.is_binary_view(dtype): + self._update_varlen(arr.cast(pa.large_binary()), b"") + elif pa.types.is_boolean(dtype): + self.streams["values"].update(pc.fill_null(arr, False).to_numpy(zero_copy_only=False).astype(np.uint8)) + elif pa.types.is_map(dtype): + entries = pa.struct([pa.field("key", dtype.key_type), pa.field("value", dtype.item_type)]) + self._update_list(arr.cast(pa.list_(entries)), nulls) # a map is a list of (key, value) entries + elif _is_list(dtype): + self._update_list(arr, nulls) + elif pa.types.is_struct(dtype): + for child, values in zip(self.children, arr.flatten()): + child.update(values) + elif _fixed_width(dtype): + self._update_fixed(arr, nulls, dtype.bit_width // 8) + else: + # Anything not covered above (unions, extension types): hash the Python values. Slow, and correct. + self.streams["values"].update(repr(arr.to_pylist()).encode()) + + def _update_varlen(self, arr: pa.Array, empty: bytes) -> None: + filled = pc.fill_null(arr, empty.decode() if pa.types.is_large_string(arr.type) else empty) + lengths = pc.binary_length(filled).to_numpy(zero_copy_only=False).astype(np.int64) + self.streams["lengths"].update(lengths.tobytes()) + if not len(filled): + return + offsets = np.frombuffer(filled.buffers()[1], dtype=np.int64)[filled.offset : filled.offset + len(filled) + 1] + data = filled.buffers()[2] + if data is not None and int(offsets[-1]) > int(offsets[0]): + self.streams["values"].update(memoryview(data)[int(offsets[0]) : int(offsets[-1])]) + + def _update_fixed(self, arr: pa.Array, nulls: np.ndarray, width: int) -> None: + if not len(arr): + return + raw = np.frombuffer(arr.buffers()[1], dtype=np.uint8)[arr.offset * width : (arr.offset + len(arr)) * width] + if arr.null_count: + raw = raw.reshape(len(arr), width).copy() + raw[nulls] = 0 # a null slot may hold anything; zero it + self.streams["values"].update(raw.tobytes() if arr.null_count else memoryview(raw)) + + def _update_list(self, arr: pa.Array, nulls: np.ndarray) -> None: + if pa.types.is_fixed_size_list(arr.type): + lengths = np.full(len(arr), arr.type.list_size, dtype=np.int64) + else: + lengths = np.diff(np.asarray(arr.offsets.to_numpy(zero_copy_only=False), dtype=np.int64)) + lengths[nulls] = 0 # a null list may be backed by a non-empty one, and flatten() drops those children + self.streams["lengths"].update(lengths.tobytes()) + self.children[0].update(arr.flatten()) + + def digest(self) -> bytes: + top = hashlib.sha256() + for stream in _STREAMS: + top.update(self.streams[stream].digest()) + for child in self.children: + top.update(child.digest()) + return top.digest() + + +def _fixed_width(dtype: pa.DataType) -> bool: + return isinstance(dtype, pa.FixedSizeBinaryType) or ( + pa.types.is_primitive(dtype) and dtype.bit_width % 8 == 0 and not pa.types.is_boolean(dtype) + ) or pa.types.is_decimal(dtype) or pa.types.is_interval(dtype) + + +class _FileHasher: + def __init__(self, schema: pa.Schema): + self.schema = schema + self.columns = [_ColumnHasher(f.name, f.type) for f in schema] + self.rows = 0 + + def update(self, batch: pa.RecordBatch | pa.Table) -> None: + for hasher, column in zip(self.columns, batch.columns): + hasher.update(column) + self.rows += batch.num_rows + + def digests(self) -> tuple[str, dict[str, str]]: + per_column = {f.name: h.digest() for f, h in zip(self.schema, self.columns)} + top = hashlib.sha256(str(self.rows).encode()) + for digest in per_column.values(): + top.update(digest) + return f"{ALGORITHM}:{top.hexdigest()}", {name: d.hex() for name, d in per_column.items()} + + +def digests_of_batches(batches, schema: pa.Schema) -> tuple[str, dict[str, str]]: + """``(file digest, {column: digest})`` of a record-batch stream, in the order the batches arrive.""" + hasher = _FileHasher(schema) + for batch in batches: + hasher.update(batch) + return hasher.digests() + + +def file_digests(path: Path) -> tuple[str, dict[str, str]]: + """``(file digest, {column: digest})`` of a parquet file, read back.""" + pf = pq.ParquetFile(path) + return digests_of_batches(pf.iter_batches(batch_size=_READ_BATCH_ROWS), pf.schema_arrow) + + +def content_digest(path: Path) -> str: + """The digest ``manifest.result_digest`` records: ``arrow-sha256:``.""" + return file_digests(path)[0] + + +def column_digests(path: Path) -> dict[str, str]: + """The digest of each column of a parquet file, by name.""" + return file_digests(path)[1] diff --git a/src/tallyman_xorq/io.py b/src/tallyman_xorq/io.py index e32e6d3..9263aff 100644 --- a/src/tallyman_xorq/io.py +++ b/src/tallyman_xorq/io.py @@ -12,6 +12,7 @@ from pathlib import Path from tallyman_core import data_dir, entry_dir, get_alias, resolve_project +from tallyman_xorq.row_order import ROW_ORDER class ProjectDataNotFound(FileNotFoundError): @@ -46,54 +47,39 @@ def read_project_file(rel_path: str, project: str | None = None): catalog hash, no recipe, no lineage entry. Everything built on top of it flows through tracked_expr_from_alias. - Source-content identity (see tallyman_xorq.source_identity): in `cas` - mode the read goes through a content-addressed clone so the path xorq - hashes embeds the content digest; in `salt` mode the digest is recorded - for build_and_persist to mix into the entry hash; `off` is the plain - path read. + The file is never read directly (ADR-008 D2). It goes through source + identity (tallyman_xorq.source_identity): its content digest is taken, in + `cas` mode it is cloned to `data/.cas/`, and in `salt` mode + the digest is recorded for build_and_persist to mix into the entry hash. + Then an *ordered copy* of it is written under the project's `compute_cache/` + (tallyman_xorq.ordered_copy): the same rows in file order plus a last column, + `__row_order`, `0..N-1`, which pages sort by. The returned expression is a + plain read of that copy, so the entry's hash covers the source's content + (the copy's name is a function of the digest) and editing the file forks it. """ from xorq.expr.api import deferred_read_parquet + from tallyman_xorq import ordered_copy as oc from tallyman_xorq import source_identity as si proj = resolve_project(project) - if si.mode() == "cas": - recorded = _reconstructing_source_digest(proj, rel_path) - if recorded is not None: - # Reconstruction (#115): this read_project_file is re-running an already-built - # entry's recipe (expr.py), not authoring a new entry. Resolve to the FROZEN - # .cas clone the entry was built from — named by the digest in its - # manifest.sources — instead of re-digesting the live file, which may have - # been edited in place since the build. Note that same frozen digest (the - # source collector is independent of tracked_expr_from_alias's _RECONSTRUCTING-gated - # parent capture, so both fire) so a child build reconstructing this entry - # records the frozen digest and gc_cas keeps the clone alive. - # - # must_exist=False: recon_cas_path serves the frozen clone without the live - # source, so a moved/deleted source must not defeat the pinned read. This - # block MUST run before the existence-requiring project_path below. - recon_path = project_path(rel_path, proj, must_exist=False) - si.note_source(rel_path, recorded) - return deferred_read_parquet(str(si.recon_cas_path(proj, recon_path, recorded))) + reader = oc.parquet_reader() + recorded = _reconstructing_source_digest(proj, rel_path) + if recorded is not None: + # Reconstruction (#115): this read_project_file is re-running an already-built entry's recipe (expr.py), not + # authoring a new entry. Resolve to the copy of the bytes the entry was BUILT from, named by the digest in its + # manifest.sources, instead of re-digesting the live file, which may have been edited in place since. Note + # that same frozen digest so a child build reconstructing this entry records it. + si.note_source(rel_path, recorded) + return deferred_read_parquet(str(oc.existing_ordered_copy(proj, digest=recorded, reader=reader))) path = project_path(rel_path, proj) + digest = si.digest_for(proj, path) if si.mode() != "off": - digest = si.digest_for(proj, path) si.note_source(rel_path, digest) - if si.mode() == "cas": - path = si.ensure_cas_path(proj, path, digest) - return deferred_read_parquet(str(path)) - - -# Pinned polars parquet-write settings for the ordered-CSV intermediate. Held -# constant so the bytes — and therefore the snapshot digest — are reproducible -# run-to-run within an environment. A polars/library upgrade may change the -# bytes; rebuild-is-fine here per the project's single-user rule. -_CSV_PARQUET_WRITE = { - "compression": "zstd", - "compression_level": 3, - "row_group_size": 122880, - "statistics": True, -} + source = si.ensure_cas_path(proj, path, digest) if si.mode() == "cas" else path + copy = oc.ensure_ordered_copy(proj, source, digest=digest, rel=rel_path, reader=reader) + return deferred_read_parquet(str(copy)) + # ibis primitive -> polars dtype, for reading a CSV with an explicit schema. # Covers the types the MCP documents for tallyman_read_csv; an unmapped type @@ -210,7 +196,7 @@ def _normalize_schema(spec, header: list[str], *, reserved: tuple[str, ...] = () """Resolve a schema spec against the CSV *header* into a polars parse plan. *header* is the CSV's full header. *reserved* names tallyman-managed columns - (currently ``original_row_order``) that the caller must not spec: they are + (currently ``__row_order``) that the caller must not spec: they are excluded from the columns the spec has to cover, but kept in the diagnostics so an error message matches the file the user is looking at rather than a silently reserved-stripped subset. Threading ``reserved`` in — rather than @@ -374,7 +360,7 @@ def _polars_to_ibis(dt) -> str: def _suggest_schema_dsl(src: Path, scan_kwargs: dict, reserved: tuple[str, ...] = ()) -> str: """Whole-file-infer *src* and render it as the paste-ready tuple-of-tuples DSL. - *reserved* columns (tallyman-managed, e.g. ``original_row_order``) are dropped: + *reserved* columns (tallyman-managed, e.g. ``__row_order``) are dropped: they must not appear in the suggestion because the caller cannot spec them — a suggestion carrying one is un-pasteable (the totality check excludes the reserved column, so pasting it back over-counts the columns and raises). @@ -386,46 +372,8 @@ def _suggest_schema_dsl(src: Path, scan_kwargs: dict, reserved: tuple[str, ...] return repr(pairs) -def _validate_existing_row_order(src: Path, scan_kwargs: dict) -> None: - """The CSV already has an ``original_row_order`` column, colliding with the row - index tallyman would add via ``with_row_index`` (a raw polars ``DuplicateError`` - the infer ladder does not catch). Reuse the column only if it *is* the canonical - order — the contiguous ``0..N-1`` file sequence tallyman itself writes, each row - exactly one more than the last. Otherwise the name is a coincidence carrying - foreign data, so raise loudly rather than trust a non-canonical order. - """ - import polars as pl - - reserved = ( - f"tallyman_read_csv: {src.name} already has a column named 'original_row_order', which is " - "reserved for tallyman's canonical row index. {why} Rename the column in the source CSV." - ) - try: - series = ( - pl.scan_csv( - str(src), - schema_overrides={"original_row_order": pl.Int64}, - infer_schema_length=0, - **scan_kwargs, - ) - .select("original_row_order") - .collect() - .to_series() - ) - except pl.exceptions.ComputeError as exc: - raise ValueError(reserved.format(why=f"Its values are not integers ({exc}).")) from exc - n = series.len() - expected = pl.int_range(0, n, dtype=pl.Int64, eager=True) - if series.null_count() or not (series == expected).all(): - raise ValueError( - reserved.format( - why=f"Its values are not the contiguous 0..{n - 1} sequence (each row must be one more than the last)." - ) - ) - - def _materialize_ordered(src: Path, schema, scan_kwargs: dict, tmp_path: Path) -> None: - """Read *src* into a row-order-stable parquet at *tmp_path* (#143). + """Read *src* (a CSV) into a row-order-stable parquet at *tmp_path* (#143). Inference mode (no schema, or a spec with inferred columns) escalates the infer window on a parse failure — 100 -> 10k -> whole-file — because a value @@ -433,60 +381,60 @@ def _materialize_ordered(src: Path, schema, scan_kwargs: dict, tmp_path: Path) - always resolves (a mixed column falls back to string). Explicit mode (every column pinned) never escalates: a pinned type that cannot parse is the caller's mistake, so it raises with a paste-ready suggested schema. + + The last column is ``__row_order`` (ADR-008 D2). A CSV that already has a + column of that name has it overwritten, which is the right outcome for a + file tallyman exported. """ import polars as pl - # Header (names only) drives both the schema plan and the row-index collision - # check; infer_schema_length=0 reads just the header line, no type sampling. + from tallyman_xorq.ordered_copy import _WRITE + + # Header (names only) drives the schema plan; infer_schema_length=0 reads + # just the header line, no type sampling. header = list(pl.scan_csv(str(src), infer_schema_length=0, **scan_kwargs).collect_schema().names()) - has_row_order = "original_row_order" in header - # original_row_order is tallyman's reserved index — _ordered re-appends it, and - # _validate_existing_row_order (below) vets a pre-existing one. It is not the caller's - # to spec, so it is threaded as `reserved` through every header reader (the schema - # plan, the diagnostics, and the suggested-schema hint) rather than stripped ad hoc. - reserved = ("original_row_order",) if has_row_order else () - if has_row_order: - _validate_existing_row_order(src, scan_kwargs) # reuse it, or raise — never with_row_index over it + has_row_order = ROW_ORDER in header + # __row_order is tallyman's reserved index — it is not the caller's to spec, so it is threaded as `reserved` + # through every header reader (the schema plan, the diagnostics, and the suggested-schema hint) rather than + # stripped ad hoc. + reserved = (ROW_ORDER,) if has_row_order else () plan = None explicit_only = False if schema is not None: plan = _normalize_schema(schema, header, reserved=reserved) - # A spec may not PRODUCE the reserved name either: renaming a data column onto - # 'original_row_order' collides with the index _ordered re-appends (a raw polars - # DuplicateError at sink). Reject the reserved output name up front with the same - # guidance as the pre-existing-column clash. - if "original_row_order" in plan["out_names"]: + # A spec may not PRODUCE the reserved name either: renaming a data column onto __row_order collides with the + # index appended below (a raw polars DuplicateError at sink). Reject the reserved output name up front. + if ROW_ORDER in plan["out_names"]: raise ValueError( - "tallyman_read_csv: 'original_row_order' is reserved for tallyman's canonical " - "row index and cannot be a schema output column name. Rename that column to " - "something other than 'original_row_order' in the schema." + f"tallyman_read_csv: {ROW_ORDER!r} is reserved for tallyman's row index and cannot be a " + f"schema output column name. Rename that column to something other than {ROW_ORDER!r} in the schema." ) explicit_only = plan["all_pinned"] def _ordered(infer_len): if schema is None: lf = pl.scan_csv(str(src), infer_schema_length=infer_len, **scan_kwargs) - if not has_row_order: - lf = lf.with_row_index("original_row_order") - cols = [c for c in lf.collect_schema().names() if c != "original_row_order"] else: il = 0 if plan["all_pinned"] else infer_len lf = pl.scan_csv(str(src), schema_overrides=plan["overrides"], infer_schema_length=il, **scan_kwargs) - if not has_row_order: - lf = lf.with_row_index("original_row_order") - if plan["rename"]: - lf = lf.rename(plan["rename"]) - cols = [c for c in plan["out_names"] if c != "original_row_order"] - # original_row_order last and cast to int64 (a fresh with_row_index yields - # uint32; a reused in-CSV column is stripped from `cols` and re-appended here). - return lf.select([*cols, pl.col("original_row_order").cast(pl.Int64)]) + if has_row_order: + lf = lf.drop(ROW_ORDER) # overwritten by the fresh index below + lf = lf.with_row_index(ROW_ORDER) + if schema is not None and plan["rename"]: + lf = lf.rename(plan["rename"]) + if schema is None: + cols = [c for c in lf.collect_schema().names() if c != ROW_ORDER] + else: + cols = [c for c in plan["out_names"] if c != ROW_ORDER] + # __row_order last and cast to int64 (a fresh with_row_index yields uint32). + return lf.select([*cols, pl.col(ROW_ORDER).cast(pl.Int64)]) ladder = [_DEFAULT_INFER] if explicit_only else [_DEFAULT_INFER, _ESCALATED_INFER, None] last_exc = None for infer_len in ladder: try: - _ordered(infer_len).sink_parquet(str(tmp_path), **_CSV_PARQUET_WRITE) + _ordered(infer_len).sink_parquet(str(tmp_path), **_WRITE) return except pl.exceptions.ComputeError as exc: last_exc = exc @@ -496,124 +444,42 @@ def _ordered(infer_len): ) -def _csv_ordered_dir() -> Path: - """Project-independent home for the ordered-CSV intermediates. - - Keyed on the absolute CSV path (+ schema + reader options), so the same file - resolves to the same intermediate regardless of which project is active — - unlike a per-project artifacts dir, which makes a reconstruction running - under a *different* active project miss the cache and re-read the CSV (#9). - Per-test isolation still holds: ``TALLYMAN_HOME`` points at a tmp dir. - """ - from tallyman_core.paths import tallyman_home +def tallyman_read_csv(path: str, schema=None, project: str | None = None, **kwargs): + """Read a CSV into a xorq expression with a stable ``__row_order``. - out_dir = tallyman_home() / "csv_ordered" - out_dir.mkdir(parents=True, exist_ok=True) - return out_dir + Use this instead of ``xo.deferred_read_csv`` for all CSV ingests. The CSV + goes through source identity like a parquet source (``read_project_file``): + its content digest is taken, it is cloned to ``data/.cas/``, and an ordered + copy of the clone is written under the project's ``compute_cache/``. polars + reads the clone (``scan_csv -> with_row_index -> sink_parquet``), which + preserves the file's row order (unlike datafusion's parallel scan, whose row + order is nondeterministic above the repartition threshold, about 10 MB) and + numbers the rows in a last column, ``__row_order``, ``0..N-1``. See + ``plans/ADR-008-row-order-of-reads.md``. - -def _ordered_csv_key(path: str, schema, scan_kwargs: dict) -> str: - """Stable, CSV-stat-independent key for the ordered intermediate. - - Derived from the absolute path, the schema, and the reader options only — - NOT the CSV's mtime/size — so a reconstruction can resolve the intermediate - without touching (or even stat-ing) the source CSV. Freshness is a separate, - best-effort mtime sidecar (see ``_ordered_csv_parquet``). - """ - import hashlib - - kwargs_sig = repr(sorted((k, repr(v)) for k, v in scan_kwargs.items())) - raw = f"{Path(path).resolve()}|{_spec_sig(schema)}|{kwargs_sig}" - return hashlib.md5(raw.encode()).hexdigest() # noqa: S324 — cache key, not a security boundary - - -def _ordered_csv_parquet(path: str, schema, scan_kwargs: dict) -> Path: - """Materialise *path* to a row-order-stable parquet and return its path. - - The CSV is read exactly once — at first ingest (and again only if its mtime - changes). The intermediate is addressed by (absolute path, schema, reader - options), not by the CSV's stat, so a faithful reconstruction resolves it - without re-reading or even stat-ing the CSV; if the source has since been - moved or deleted the existing intermediate is still served rather than - raising (#6). A sidecar records the *source* CSV's mtime so a changed CSV - re-materialises (#8 — mtime invalidates; bytes/size are unnecessary). The - sidecar tracks the source's mtime, never a derived parquet's, so the - result-digest / snapshot bake can never bump it into a re-ingest loop. - - polars ``scan_csv -> with_row_index -> sink_parquet`` preserves source file - order (unlike datafusion's parallel scan, whose row order is nondeterministic - above the repartition threshold), so ``original_row_order`` is the true - 0..N-1 file sequence and the bytes are reproducible. See - ``plans/ADR-004-result-digest-canonical-ordering.md``. - """ - import os - import uuid - - src = Path(path) - out_dir = _csv_ordered_dir() - key = _ordered_csv_key(path, schema, scan_kwargs) - target = out_dir / f"{key}.parquet" - stamp = out_dir / f"{key}.mtime" - - if target.exists(): - try: - current = str(src.stat().st_mtime_ns) - except OSError: - return target # source gone — the baked intermediate IS the source of truth (#6) - try: - recorded = stamp.read_text() - except OSError: - recorded = None - if recorded == current: - return target # unchanged source — reuse, never re-read the CSV (#6) - # mtime changed → fall through and re-materialise (#8) - - st = src.stat() # the single point that touches the CSV; it must be present here - # Unique temp name (not a fixed {key}.parquet.tmp) so two builds writing the - # same key concurrently can't corrupt each other's partial file (#7); the - # rename is atomic and last-writer-wins on identical bytes. - tmp = out_dir / f"{key}.{uuid.uuid4().hex}.tmp" - try: - _materialize_ordered(src, schema, scan_kwargs, tmp) - os.replace(tmp, target) - finally: - tmp.unlink(missing_ok=True) - stamp.write_text(str(st.st_mtime_ns)) - return target - - -def tallyman_read_csv(path: str, schema=None, **kwargs): - """Read a CSV into a xorq expression with a stable ``original_row_order``. - - Use this instead of ``xo.deferred_read_csv`` for all CSV ingests. It adds a - 0-based ``original_row_order`` column holding the source file's true row - sequence, which serves as the canonical total order for snapshot baking — - making the entry's result digest byte-stable across builds. - - The CSV is read exactly once, at ingest: an order-preserving polars read - (``scan_csv -> with_row_index``) materialises it to a parquet intermediate, - and the returned expression is a ``deferred_read_parquet`` of that file with - a top-level ``order_by("original_row_order")``. All downstream computation - (the recipe, the snapshot bake, the result digest) runs on the parquet, never - the CSV. polars is used rather than ``ibis.row_number()`` because a bare - datafusion ``ROW_NUMBER() OVER ()`` numbers rows in the nondeterministic - arrival order of a parallel CSV scan (any file over datafusion's ~10 MB - repartition threshold), so it would not pin file order at all. See - ``plans/ADR-004-result-digest-canonical-ordering.md``. + The CSV is read exactly once, at ingest, and the returned expression is a + plain ``deferred_read_parquet`` of the copy: no sort, one row-order column. + Because the copy is named by the CSV's content and the reader options, + editing the CSV and running the same recipe gives a new content hash, and the + earlier entry keeps the rows it was built from (#168). Args: path: Absolute path to the CSV file. schema: Optional ibis schema for the columns (same as deferred_read_csv). When omitted, polars infers types. + project: Project name override (defaults to the active project). **kwargs: Forwarded to ``polars.scan_csv`` — reader options such as ``separator``, ``skip_rows``, ``null_values``, ``quote_char``, - ``has_header``, ``encoding``. They participate in the intermediate's - cache key, so changing one re-ingests. ``infer_schema_length`` and + ``has_header``, ``encoding``. They participate in the copy's key, + so changing one re-ingests. ``infer_schema_length`` and ``schema_overrides`` are managed internally (see ``_RESERVED_SCAN_KWARGS``) and rejected — they would collide with the values every internal ``scan_csv`` call already sets. """ - import xorq.api as xo + from xorq.expr.api import deferred_read_parquet + + from tallyman_xorq import ordered_copy as oc + from tallyman_xorq import source_identity as si reserved = [k for k in _RESERVED_SCAN_KWARGS if k in kwargs] if reserved: @@ -623,8 +489,29 @@ def tallyman_read_csv(path: str, schema=None, **kwargs): "(100 -> 10k -> whole-file); to pin column types pass schema= (an ibis schema, " "plain dict, or tuple-of-tuples), never schema_overrides." ) - parquet_path = _ordered_csv_parquet(path, schema, kwargs) - return xo.deferred_read_parquet(str(parquet_path)).order_by("original_row_order") + proj = resolve_project(project) + reader = oc.csv_reader(schema, kwargs) + src = Path(path) + rel = _relative_to_data(proj, src) + recorded = _reconstructing_source_digest(proj, rel) + if recorded is not None: + si.note_source(rel, recorded) + return deferred_read_parquet(str(oc.existing_ordered_copy(proj, digest=recorded, reader=reader))) + digest = si.digest_for(proj, src) + if si.mode() != "off": + si.note_source(rel, digest) + source = si.ensure_cas_path(proj, src, digest) if si.mode() == "cas" else src + copy = oc.ensure_ordered_copy(proj, source, digest=digest, rel=rel, reader=reader) + return deferred_read_parquet(str(copy)) + + +def _relative_to_data(proj: str, path: Path) -> str: + """The name a source is recorded under in ``manifest.sources``: relative to the project's data dir when it sits + there, the absolute path otherwise (a CSV can live anywhere).""" + try: + return str(path.resolve().relative_to(data_dir(proj).resolve())) + except ValueError: + return str(path) def _reconstructing_source_digest(proj: str, rel_path: str) -> str | None: @@ -647,25 +534,29 @@ def _reconstructing_source_digest(proj: str, rel_path: str) -> str | None: return sources.get(rel_path) -def _note_parent_sources(proj: str, content_hash: str) -> None: - """Fold the parent's recorded source digests into the current build's collector. +def _note_parent_records(proj: str, content_hash: str) -> None: + """Fold the parent's recorded source digests, and its ordered copies when its graph is inlined, into the build. - A child's build inlines its parent's frozen graph — CAS clone paths included - — so the child's manifest must record those digests too: ``manifest.sources`` - is the closure record ``gc_cas`` walks to keep clones alive, and the child - must keep its own leaves alive even if the parent entry is later evicted. - ``note_source`` is a no-op outside a build's collect window, so this costs - nothing on a plain read. + The child's ``manifest.sources`` is the closure record ``gc_cas`` walks to keep clones alive, so it carries every + digest its parent recorded, and the child keeps its own leaves alive even if the parent entry is later evicted. + A CHEAP parent's graph is inlined into the child's build (a worthy parent is a bare read of its snapshot), so the + child reads that parent's ordered copies directly and needs the records that let ``ensure_materialized`` make a + deleted one again (ADR-007 D13). ``note_source`` and ``note_ordered_copy`` are no-ops outside a build's collect + window, so this costs nothing on a plain read. """ from tallyman_core import read_manifest + from tallyman_xorq import ordered_copy as oc from tallyman_xorq import source_identity as si try: - sources = read_manifest(entry_dir(proj, content_hash)).sources or {} + manifest = read_manifest(entry_dir(proj, content_hash)) except (OSError, ValueError): return - for rel_path, digest in sources.items(): + for rel_path, digest in (manifest.sources or {}).items(): si.note_source(rel_path, digest) + if manifest.cache_worthy is False: + for key, record in (manifest.ordered_copies or {}).items(): + oc.note_ordered_copy(key, record) def tracked_expr_from_alias(alias: str, project: str | None = None): @@ -680,13 +571,14 @@ def tracked_expr_from_alias(alias: str, project: str | None = None): current head entry. Pass the alias name as it appears in ``catalog_list``. To read by hash without tracking, use ``pinned_expr_from_alias``. - Returns the parent entry's *frozen build graph* on the in-process default - backend (``entry_graph_expr``, ADR D4): parents are inlined by value — - content-pinned sources, and an expensive parent's baked ``CachedNode`` - travels along — so the child's own build becomes self-contained and - self-healing. When this alias is revised, recalc mints a new version of the - child expression; the child entry built *now* stays bound to the parent - revision recorded at build time forever. + Returns the parent entry's result on the in-process default backend + (``cached_result_expr``, ADR-007 D3): for a worthy parent a bare read of + its snapshot, whose path contains the parent's content hash, so the child's + identity is a function of the parent's; for a cheap parent the parent's own + frozen graph. The snapshot is made to exist first (``ensure_materialized``), + since a child cannot be built over a file that is missing. When this alias is + revised, recalc mints a new version of the child expression; the child entry + built *now* stays bound to the parent revision recorded at build time forever. The parent edge is suppressed during reconstruction (when ``_RECONSTRUCTING`` is True — the structural-nondeterminism diagnostic re-running a recipe) so a @@ -698,7 +590,7 @@ def tracked_expr_from_alias(alias: str, project: str | None = None): content hash — pass hashes to pinned_expr_from_alias instead. project: Project name override (defaults to active TALLYMAN_PROJECT). """ - from tallyman_xorq.result_cache import _RECONSTRUCTING, _resolve_noncyclic_hash, entry_graph_expr + from tallyman_xorq.result_cache import _RECONSTRUCTING, _resolve_noncyclic_hash, cached_result_expr proj = resolve_project(project) if entry_dir(proj, alias).exists(): @@ -714,8 +606,8 @@ def tracked_expr_from_alias(alias: str, project: str | None = None): from tallyman_xorq import parent_capture as pc pc.note_parent(content_hash, ref=alias, follow=True) - _note_parent_sources(proj, content_hash) - return entry_graph_expr(proj, content_hash) + _note_parent_records(proj, content_hash) + return cached_result_expr(proj, content_hash) def pinned_expr_from_alias(ref: str, project: str | None = None): @@ -740,7 +632,7 @@ def pinned_expr_from_alias(ref: str, project: str | None = None): project: Project name override (defaults to active TALLYMAN_PROJECT). """ from tallyman_core.aliases import VERSION_REF_RE, history_for, resolve_version_ref - from tallyman_xorq.result_cache import _RECONSTRUCTING, _resolve_noncyclic_hash, entry_graph_expr + from tallyman_xorq.result_cache import _RECONSTRUCTING, _resolve_noncyclic_hash, cached_result_expr proj = resolve_project(project) if entry_dir(proj, ref).exists(): @@ -770,5 +662,5 @@ def pinned_expr_from_alias(ref: str, project: str | None = None): from tallyman_xorq import parent_capture as pc pc.note_parent(content_hash, ref=ref, follow=False) - _note_parent_sources(proj, content_hash) - return entry_graph_expr(proj, content_hash) + _note_parent_records(proj, content_hash) + return cached_result_expr(proj, content_hash) diff --git a/src/tallyman_xorq/materialize.py b/src/tallyman_xorq/materialize.py new file mode 100644 index 0000000..ec47542 --- /dev/null +++ b/src/tallyman_xorq/materialize.py @@ -0,0 +1,306 @@ +"""Tallyman owns result materialization (ADR-007 D4 and D5, ADR-009 D1, D3 and D6). + +A **worthy** entry has a **snapshot**: a parquet file, ``compute_cache/result_cache/.parquet``, written +once when the entry is created and read by everything after. Nothing else writes it. xorq's cache nodes are not in +any build, so no other process runs an entry's expensive computation, writes result files or repairs tallyman's cache +(ADR-007, the governing rule). + +``materialize`` is that one writer. The build calls it and so does every heal, so the contract's "result bytes are +manufactured exactly once" has one routine to hold to. It runs the entry's frozen build on a single-partition +connection, so a float aggregate merges its partial sums in one order (ADR-009 D1), streams the rows through a writer +that fixes the layout of the file (ADR-009 D3) and numbers them in a last column, ``__row_order`` (ADR-008 D2), and +returns the content digest of the file it wrote, read back (ADR-009 D2). + +``ensure_materialized`` is the one entry point that makes files exist: every snapshot, ordered copy and clone an +entry's plan reads is on disk before anything executes (ADR-007 D5). +""" + +from __future__ import annotations + +import logging +import os +import uuid +from dataclasses import dataclass, field +from pathlib import Path + +import numpy as np +import pyarrow as pa +import pyarrow.parquet as pq + +from tallyman_xorq.row_order import ROW_ORDER, ROW_ORDER_RIGHT + +perf_log = logging.getLogger("tallyman.perf") + +# The snapshot format (ADR-009 D3). The row-group size decides the batch boundaries an entry built on this file sees, +# and an ungrouped float total depends on them (#187), so the row-group size and the materialization connection's batch +# size are part of the reproducibility contract: changing either is a corpus rebuild, and the version below stands for +# both (and for the layout of the ordered copies of sources, ``ordered_copy.ORDERED_COPY_ROW_GROUP_ROWS``). +SNAPSHOT_ROW_GROUP_ROWS = 1_048_576 +SNAPSHOT_BATCH_SIZE = 8192 +SNAPSHOT_FORMAT_VERSION = 1 + +_PARQUET_OPTIONS = { + "compression": "zstd", + "compression_level": 3, + "version": "2.6", + "data_page_version": "1.0", + "write_statistics": True, + "write_page_index": True, # what lets a page be fetched as a range of __row_order without decoding a row group +} + +RESULT_CACHE_DIRNAME = "result_cache" + + +def snapshots_dir(project: str) -> Path: + from tallyman_core.paths import compute_cache_dir + + return compute_cache_dir(project) / RESULT_CACHE_DIRNAME + + +def snapshot_path(project: str, content_hash: str) -> Path: + """Where the entry's snapshot lives: a function of the content hash and nothing else (ADR-007 D2).""" + return snapshots_dir(project) / f"{content_hash}.parquet" + + +def engine_versions() -> dict[str, str]: + """The versions of what decides an entry's bytes, recorded at build so a mismatch at a heal can be attributed.""" + from importlib.metadata import PackageNotFoundError, version + + out = {} + for key, dist in (("xorq", "xorq"), ("xorq_datafusion", "xorq-datafusion"), ("pyarrow", "pyarrow")): + try: + out[key] = version(dist) + except PackageNotFoundError: + out[key] = "unknown" + return out + + +def single_partition_backend(): + """A fresh datafusion backend that runs every query as one stream, with an explicit batch size (ADR-009 D1). + + With one partition the partial results of an aggregate are merged in one order, so a float ``SUM`` or ``AVG`` is + bit-stable on any machine, and a plan that keeps rows streams them in the parent file's order. It is a separate + connection from the default backend that serves page reads: a long materialization must not share a context with + them, and the default keeps the machine's core count of partitions. + """ + from tallyman_xorq.backend import connect + + con = connect() + con.raw_sql("SET datafusion.execution.target_partitions = 1") + con.raw_sql(f"SET datafusion.execution.batch_size = {SNAPSHOT_BATCH_SIZE}") + return con + + +@dataclass +class Materialized: + """What one materialization wrote.""" + + path: Path + digest: str # the content digest of the file as read back, ``arrow-sha256:`` + row_count: int + schema: pa.Schema # read from the written file (parquet changes some types: timestamp[s] comes back [ms]) + # Only when ``check_reproducible``: whether the second run wrote the same digest, and the columns that differed. + reproducible: bool | None = None + differing_columns: list[str] = field(default_factory=list) + + +def _stream_to_parquet(expr, dest: Path) -> tuple[int, pa.Schema]: + """Write the rows of *expr* to *dest* in the snapshot format, numbering them in a last ``__row_order`` column. + + The writer drops an inherited ``__row_order`` (each materialization overwrites it with positions in its own file) + and ibis's ``__row_order_right`` (a join's leftover copy of the right side's, which would collide in the next join, + ADR-008 D6). It regroups the stream into row groups of ``SNAPSHOT_ROW_GROUP_ROWS`` and combines each into + contiguous arrays, so the file does not depend on how the engine batched the rows. Memory is bounded by one row + group. + """ + reader = expr.to_pyarrow_batches() + kept = [f for f in reader.schema if f.name not in (ROW_ORDER, ROW_ORDER_RIGHT)] + out_schema = pa.schema([*kept, pa.field(ROW_ORDER, pa.int64())]) + names = [f.name for f in kept] + written = 0 + + with pq.ParquetWriter(dest, out_schema, **_PARQUET_OPTIONS) as writer: + + def flush(table: pa.Table) -> None: + nonlocal written + n = table.num_rows + numbered = table.select(names).append_column( + pa.field(ROW_ORDER, pa.int64()), pa.array(np.arange(written, written + n, dtype=np.int64)) + ) + if not numbered.schema.equals(out_schema, check_metadata=False): + numbered = numbered.cast(out_schema) + writer.write_table(numbered.combine_chunks(), row_group_size=SNAPSHOT_ROW_GROUP_ROWS) + written += n + + pending: list[pa.RecordBatch] = [] + pending_rows = 0 + for batch in reader: + if not batch.num_rows: + continue + pending.append(batch) + pending_rows += batch.num_rows + while pending_rows >= SNAPSHOT_ROW_GROUP_ROWS: + table = pa.Table.from_batches(pending) + flush(table.slice(0, SNAPSHOT_ROW_GROUP_ROWS)) + tail = table.slice(SNAPSHOT_ROW_GROUP_ROWS) + pending, pending_rows = tail.to_batches(), tail.num_rows + if pending_rows: + flush(pa.Table.from_batches(pending)) + return written, out_schema + + +def _run_once(project: str, content_hash: str, dest: Path) -> int: + """Run the entry's frozen build on a single-partition connection and write the result to *dest*.""" + from tallyman_xorq.result_cache import load_entry_expr, rebind_onto + + expr = rebind_onto(load_entry_expr(project, content_hash), single_partition_backend()) + rows, _ = _stream_to_parquet(expr, dest) + return rows + + +def _temp_beside(path: Path) -> Path: + # A unique temp name in the destination directory: same filesystem for the replace, and no two writers share one. + return path.with_name(f".{path.stem}.{uuid.uuid4().hex}.tmp") + + +def materialize(project: str, content_hash: str, *, check_reproducible: bool = False) -> Materialized: + """Run the entry's build and write its snapshot; return what was written (ADR-007 D4). + + It always runs the query and replaces whatever is at the path, so an entry that is added again after a reset is + honest and the reproducibility check below has something to compare. It takes the project's write lock, and the + file only ever changes by an atomic replace of a complete one. It does not read the manifest: a create calls it + before the manifest exists. + + With ``check_reproducible`` (what a create does, ADR-009 D6) it runs the query twice through the same writer and + compares the two content digests. The second file is discarded. When they differ the entry is not reproducible, + and ``differing_columns`` names the columns whose digests differ. A heal runs the query once. + """ + from tallyman_core.catalog_state import project_lock + from tallyman_xorq.digest import file_digests + + dest = snapshot_path(project, content_hash) + with project_lock(project): + dest.parent.mkdir(parents=True, exist_ok=True) + tmp = _temp_beside(dest) + try: + rows = _run_once(project, content_hash, tmp) + digest, columns = file_digests(tmp) + reproducible: bool | None = None + differing: list[str] = [] + if check_reproducible: + second = _temp_beside(dest) + try: + _run_once(project, content_hash, second) + digest_again, columns_again = file_digests(second) + finally: + second.unlink(missing_ok=True) + reproducible = digest_again == digest + if not reproducible: + differing = [c for c in columns if columns[c] != columns_again.get(c)] or list(columns) + os.replace(tmp, dest) + finally: + tmp.unlink(missing_ok=True) + return Materialized(dest, digest, rows, pq.read_schema(dest), reproducible, differing) + + +# --------------------------------------------------------------------------- +# ensure_materialized +# --------------------------------------------------------------------------- + + +def _heal(project: str, content_hash: str) -> None: + """Re-create the entry's snapshot, verified against the manifest's ``result_digest`` before it is served.""" + from tallyman_core.catalog_state import project_lock + from tallyman_xorq.result_cache import _verify_self_heal + + with project_lock(project): + if snapshot_path(project, content_hash).exists(): # a peer thread or process healed it while we waited + return + result = materialize(project, content_hash) + perf_log.debug("ensure_materialized healed %s", content_hash) + _verify_self_heal(project, content_hash, result.digest) + + +def _recreate(project: str, owner_hash: str, path: Path) -> None: + """Make a missing file the owner's plan reads again, by the rule for its class (ADR-007 D13).""" + from tallyman_xorq import ordered_copy as oc + from tallyman_xorq.build import BuildError + + if path.parent == snapshots_dir(project): + from tallyman_core.paths import entry_dir + + parent = path.stem + if not entry_dir(project, parent).is_dir(): + raise BuildError( + f"entry {owner_hash} in {project!r} reads the snapshot of entry {parent}, which is not in this " + "catalog, so it cannot be made again" + ) + ensure_materialized(project, parent) + elif oc.is_ordered_copy_path(project, path): + oc.recreate_ordered_copy(project, owner_hash, path) + else: + raise BuildError( + f"entry {owner_hash} in {project!r} reads {path}, which tallyman did not write and cannot make again" + ) + if not path.exists(): + raise BuildError(f"entry {owner_hash} in {project!r}: {path} is still missing after it was made again") + + +def _ensure(project: str, content_hash: str) -> bool: + """``ensure_materialized``, returning whether the entry is worthy (the caller usually needs to know).""" + from tallyman_xorq.result_cache import _resolve_result_plan, cache_worthy + + worthy = cache_worthy(project, content_hash) + if worthy and snapshot_path(project, content_hash).exists(): + return True + plan = _resolve_result_plan(project, content_hash) + for path in plan.reads: + if not path.exists(): + _recreate(project, content_hash, path) + if worthy and not snapshot_path(project, content_hash).exists(): + _heal(project, content_hash) + return worthy + + +def ensure_materialized(project: str, content_hash: str) -> None: + """Guarantee that every file an entry's plan reads, and its own snapshot, is on disk (ADR-007 D5). + + 1. A worthy entry whose snapshot exists is done, and no build is loaded. + 2. Otherwise load the entry's build (the plan is kept in the existing LRU) and collect every file its ``Read`` + nodes point at. + 3. Re-create each that is missing by the rule for its class: a snapshot by recursing on the hash in its file name, + an ordered copy of a source from its clone, a clone from the live source while the bytes still match. + 4. If the entry is worthy, heal its own snapshot and verify it. + + Every caller that composes or executes an entry goes through here, so nothing ever runs over a file that is + missing. When nothing can make a file again, the error names the source file. + """ + _ensure(project, content_hash) + + +def pinned_reason(project: str, content_hash: str) -> str | None: + """Why the entry's snapshot must not be deleted, or None when it may be (ADR-009 D6, ADR-007 D12). + + A snapshot is pinned when it cannot be made again faithfully: the recipe is not reproducible (two runs at create + time gave different digests), or a heal already produced different rows than were built. The Cache page's delete + leaves such a file alone and says why. ``compute_cache/`` as a whole is still deletable by definition. + """ + from tallyman_core import read_manifest + from tallyman_core.errors import list_errors + from tallyman_core.paths import entry_dir + + try: + manifest = read_manifest(entry_dir(project, content_hash)) + except (OSError, ValueError): + manifest = None + if manifest is not None and manifest.reproducible is False: + columns = ", ".join(manifest.nonreproducible_columns or []) + return ( + "this entry's query is not reproducible (two runs at create time gave different results" + + (f" in {columns}" if columns else "") + + "), so its snapshot cannot be re-created faithfully and is kept" + ) + for record in list_errors(project, limit=1_000_000_000): + if record.get("code") == "unfaithful_heal" and record.get("hash") == content_hash: + return "a heal of this snapshot produced different rows than were built, so it is kept" + return None diff --git a/src/tallyman_xorq/ordered_copy.py b/src/tallyman_xorq/ordered_copy.py new file mode 100644 index 0000000..03ec530 --- /dev/null +++ b/src/tallyman_xorq/ordered_copy.py @@ -0,0 +1,322 @@ +"""Ordered copies of sources (ADR-008 D2, ADR-007 D13): the one way a file enters a recipe. + +A source (a parquet file or a CSV under the project) is read through its content-addressed clone (``data/.cas``), and +polars writes a parquet copy of it, in file order, with one more column at the end: ``__row_order``, ``0..N-1``. That +copy is what a recipe reads. Nothing reads the source or the clone directly, so: + +- every file tallyman reads carries the column that pages sort by; +- the copy is keyed by the source's content digest and the reader options, so editing a source and running the same + recipe forks the entry's hash (#168) and the old entry keeps the rows it was built from; +- the copy is cache (ADR-007 D13). It lives under ``compute_cache/``, so a project that is packed, cloned or emptied + loses it, and ``ensure_materialized`` makes it again from the clone with the reader options the manifest records, + then checks it against the content digest recorded when it was first written. + +The layout of the copy is part of the reproducibility contract (ADR-009 D3, #187): an ungrouped float total depends on +the row-group boundaries of the file it reads, so the row-group size below is frozen with the snapshot format version. +""" + +from __future__ import annotations + +import contextvars +import hashlib +import json +import logging +import os +import uuid +from pathlib import Path + +from tallyman_xorq.row_order import ROW_ORDER + +perf_log = logging.getLogger("tallyman.perf") + +# Pinned polars parquet-write settings for an ordered copy. Held constant so the layout, and therefore any float total +# computed straight from a source, is reproducible. Changing one is a corpus rebuild (SNAPSHOT_FORMAT_VERSION). +ORDERED_COPY_ROW_GROUP_ROWS = 122_880 +_WRITE = { + "compression": "zstd", + "compression_level": 3, + "row_group_size": ORDERED_COPY_ROW_GROUP_ROWS, + "statistics": True, +} + +ORDERED_COPIES_DIRNAME = "ordered_sources" + + +class SourceUnavailable(FileNotFoundError): + """A file the entry needs cannot be made again: its clone is gone and the live source is not the bytes it read.""" + + +def ordered_copies_dir(project: str) -> Path: + from tallyman_core.paths import compute_cache_dir + + return compute_cache_dir(project) / ORDERED_COPIES_DIRNAME + + +def is_ordered_copy_path(project: str, path: Path) -> bool: + return Path(path).parent == ordered_copies_dir(project) + + +# --------------------------------------------------------------------------- +# readers: what the manifest records so a copy can be made again +# --------------------------------------------------------------------------- + + +def parquet_reader() -> dict: + return {"kind": "parquet"} + + +def csv_reader(schema, scan_kwargs: dict) -> dict: + """The reader options of a CSV source, in a form that goes into JSON and can be replayed. + + ``lossless`` is False when a ``scan_csv`` option does not survive JSON (the copy is then keyed correctly but cannot + be made again from the manifest alone, and the error says so). + """ + try: + json.dumps(scan_kwargs, sort_keys=True) + lossless = True + except TypeError: + lossless = False + return { + "kind": "csv", + "schema": _spec_to_json(schema), + "scan_kwargs": json.loads(json.dumps(scan_kwargs, sort_keys=True, default=repr)), + "lossless": lossless, + } + + +def _spec_to_json(spec): + if spec is None: + return None + if isinstance(spec, (tuple, list)): + return {"form": "positional", "cells": [[str(n), str(d)] for n, d in spec]} + if hasattr(spec, "names") and hasattr(spec, "types"): + return {"form": "named", "cells": [[str(n), str(t)] for n, t in zip(spec.names, spec.types)]} + if isinstance(spec, dict): + return {"form": "named", "cells": [[str(k), str(v)] for k, v in spec.items()]} + raise ValueError(f"tallyman_read_csv: unsupported schema spec type {type(spec).__name__!r}.") + + +def _spec_from_json(doc): + if doc is None: + return None + cells = [(n, d) for n, d in doc["cells"]] + return tuple(cells) if doc["form"] == "positional" else dict(cells) + + +def _reader_signature(reader: dict) -> str: + return json.dumps({k: v for k, v in reader.items() if k != "lossless"}, sort_keys=True) + + +def copy_key(digest: str, reader: dict) -> str: + """The copy's file stem: a function of the source's content and the reader options, nothing else.""" + return hashlib.md5(f"{digest}|{_reader_signature(reader)}".encode()).hexdigest() # noqa: S324 — a cache key + + +# --------------------------------------------------------------------------- +# collecting the records a build needs for its manifest +# --------------------------------------------------------------------------- + +_collector: contextvars.ContextVar[dict[str, dict] | None] = contextvars.ContextVar( + "tallyman_ordered_copy_collector", default=None +) + + +def begin_collect() -> contextvars.Token: + return _collector.set({}) + + +def note_ordered_copy(key: str, record: dict) -> None: + bag = _collector.get() + if bag is not None: + bag[key] = record + + +def end_collect(token: contextvars.Token) -> dict[str, dict]: + bag = _collector.get() or {} + _collector.reset(token) + return dict(bag) + + +# --------------------------------------------------------------------------- +# writing +# --------------------------------------------------------------------------- + + +def _write_parquet_copy(src: Path, dest: Path) -> None: + import polars as pl + + lf = pl.scan_parquet(str(src)) + columns = [c for c in lf.collect_schema().names() if c != ROW_ORDER] # an existing __row_order is overwritten + lf.select(columns).with_row_index(ROW_ORDER).select([*columns, pl.col(ROW_ORDER).cast(pl.Int64)]).sink_parquet( + str(dest), **_WRITE + ) + + +def _write_copy(src: Path, reader: dict, dest: Path) -> None: + if reader["kind"] == "parquet": + _write_parquet_copy(src, dest) + return + from tallyman_xorq.io import _materialize_ordered + + _materialize_ordered(src, _spec_from_json(reader["schema"]), dict(reader["scan_kwargs"]), dest) + + +def _digest_sidecar(path: Path) -> Path: + return path.with_suffix(".digest") + + +def _content_digest_of(path: Path) -> str: + """The copy's content digest: the sidecar written with it, or a read-back when the sidecar is gone.""" + from tallyman_xorq.digest import content_digest + + sidecar = _digest_sidecar(path) + try: + return sidecar.read_text().strip() + except OSError: + digest = content_digest(path) + try: + sidecar.write_text(digest) + except OSError: + pass + return digest + + +def _write_atomically(src: Path, reader: dict, target: Path) -> str: + """Write the copy to a unique temp name beside its destination, replace, and return its content digest.""" + from tallyman_xorq.digest import content_digest + + target.parent.mkdir(parents=True, exist_ok=True) + tmp = target.with_name(f"{target.stem}.{uuid.uuid4().hex}.tmp") + try: + _write_copy(src, reader, tmp) + os.replace(tmp, target) + finally: + tmp.unlink(missing_ok=True) + digest = content_digest(target) + _digest_sidecar(target).write_text(digest) + return digest + + +def ensure_ordered_copy(project: str, source: Path, *, digest: str, rel: str, reader: dict) -> Path: + """The ordered copy of *source* (whose content digest is *digest*), written if it is not on disk yet. + + *source* is the file polars reads: the clone in cas mode, the live file otherwise. The copy is recorded for the + build in progress, so its manifest can make it again. + """ + from tallyman_core.catalog_state import project_lock + + key = copy_key(digest, reader) + target = ordered_copies_dir(project) / f"{key}.parquet" + if not target.exists(): + with project_lock(project): + if not target.exists(): # a peer may have written it while we waited + _write_atomically(source, reader, target) + note_ordered_copy( + key, + { + "source": rel, + "digest": digest, + "suffix": Path(source).suffix, + "reader": reader, + "content_digest": _content_digest_of(target), + }, + ) + return target + + +def existing_ordered_copy(project: str, *, digest: str, reader: dict) -> Path: + """The path a copy of a source with this digest and reader has, without touching the source (reconstruction).""" + return ordered_copies_dir(project) / f"{copy_key(digest, reader)}.parquet" + + +# --------------------------------------------------------------------------- +# making one again +# --------------------------------------------------------------------------- + + +def _live_source(project: str, source: str) -> Path: + from tallyman_core.paths import data_dir + + p = Path(source) + return p if p.is_absolute() else data_dir(project) / source + + +def _clone_or_live(project: str, record: dict) -> Path: + """The file to re-read a source from: its clone, made again from the live source if that still has the bytes.""" + from tallyman_core.paths import data_dir + from tallyman_xorq import source_identity as si + + digest, suffix = record["digest"], record.get("suffix", "") + clone = data_dir(project) / ".cas" / f"{digest}{suffix}" + if clone.exists(): + return clone + live = _live_source(project, record["source"]) + if live.is_file() and si._digest_file(live) == digest: + return si.ensure_cas_path(project, live, digest) if si.mode() == "cas" else live + raise SourceUnavailable( + f"the source file {record['source']!r} cannot be read the way it was when the entry was built: its " + f"content-addressed clone ({digest}{suffix}) is gone and the file on disk no longer has those bytes. The " + "rows the entry was built from are unrecoverable; rebuild the entry from the current file." + ) + + +def recreate_ordered_copy(project: str, owner_hash: str, path: Path) -> None: + """Make a deleted ordered copy again for the entry *owner_hash*, and check it against its recorded digest. + + Uses the reader options the entry's manifest recorded. A copy whose digest differs from the one recorded when it + was first written is served all the same, but never silently (ADR-007 D5): a durable error record and a warning. + """ + from tallyman_core.catalog_state import project_lock + from tallyman_core.manifest import read_manifest + from tallyman_core.paths import entry_dir + + key = Path(path).stem + record = ((read_manifest(entry_dir(project, owner_hash)).ordered_copies) or {}).get(key) + if record is None: + raise SourceUnavailable( + f"entry {owner_hash} reads {path.name} but its manifest has no record of the source it was made from, so " + "it cannot be made again; rebuild the entry" + ) + reader = record["reader"] + if reader.get("lossless") is False: + raise SourceUnavailable( + f"the reader options of {record['source']!r} were not JSON-serializable, so {path.name} cannot be made " + "again from the manifest; rebuild the entry" + ) + with project_lock(project): + if path.exists(): + return + src = _clone_or_live(project, record) + digest = _write_atomically(src, reader, path) + if digest != record["content_digest"]: + message = ( + f"the ordered copy of {record['source']!r} was made again with content digest {digest}, not the " + f"{record['content_digest']} recorded when it was first written (a change in polars, or a source that is " + "not what it was)" + ) + perf_log.warning("recreate_ordered_copy %s: %s", key, message) + try: + from tallyman_core.errors import record_error + + record_error(project, code="unfaithful_ordered_copy", message=message, hash=owner_hash) + except Exception: + perf_log.debug("unfaithful ordered copy record failed for %s", key, exc_info=True) + + +def describe_read(project: str, path: Path, records: dict[str, dict] | None = None) -> str: + """A phrase naming what a Read of *path* is, for an error that has to say what an entry reads. + + ``records`` are the ordered copies the build in progress collected, so a copy is named by the source it was made + from and not by its key. + """ + from tallyman_core.aliases import alias_for_hash + from tallyman_xorq.materialize import snapshots_dir + + path = Path(path) + if path.parent == snapshots_dir(project): + alias = alias_for_hash(project, path.stem) + return f"entry {path.stem}" + (f" ({alias})" if alias else "") + record = (records or {}).get(path.stem) + if record is not None: + return f"the source {record['source']}" + return f"the file {path.name}" diff --git a/src/tallyman_xorq/portable.py b/src/tallyman_xorq/portable.py index c466f9d..d0702b5 100644 --- a/src/tallyman_xorq/portable.py +++ b/src/tallyman_xorq/portable.py @@ -17,6 +17,11 @@ `xorq_build/database_tables/` reads that source from the expanded dir at execute time — which happens long after load. Tearing the dir down first makes the read silently yield zero rows. + +A build holds no cache node (ADR-007 D1), so nothing in it needs a cache directory +at load time; the only paths in a child's `expr.yaml` that name project files are +the literal paths of its parent's snapshot and of ordered copies of sources, which +the `${TALLYMAN_PROJECT_ROOT}` placeholder covers. """ from __future__ import annotations @@ -75,37 +80,6 @@ def expand_into_dir(build_dir: Path, project_root: Path, target: Path) -> None: (target / item.name).write_text(text.replace(PLACEHOLDER, project_root_str)) -def rewrite_cache_dirs(expr, cache_dir: Path): - """Point every ``CachedNode``'s storage at *cache_dir* — nested nodes included. - - xorq's own ``load_expr(cache_dir=…)`` rewrite (``ExprLoader.replace_base_path``) - walks the graph with vendored ibis's ``.replace``, which does not descend into - opaque ``Expr``-typed fields — so a cache node nested inside another cache - node's ``parent`` (a chained parent's cache traveling in a child's build, ADR - D4) keeps whatever base path the build serialized. This deep variant runs the - same storage rewrite through xorq's ``replace_nodes``, whose traversal does - descend ``CachedNode.parent``, so every cache node in the closure lands in the - project's compute cache regardless of nesting depth or where the build was - written. - """ - from attr import evolve - from xorq.common.utils.graph_utils import replace_nodes - from xorq.expr.relations import CachedNode - - cache_dir = Path(cache_dir) - - def replacer(node, kwargs): - if isinstance(node, CachedNode): - storage = getattr(node.cache, "storage", None) - if storage is not None and Path(str(storage.base_path)) != cache_dir: - evolved = evolve(node.cache, storage=evolve(storage, base_path=cache_dir)) - return node.__recreate__(dict(zip(node.__argnames__, node.__args__)) | {"cache": evolved}) - return node - return node.__recreate__(kwargs) if kwargs else node - - return replace_nodes(replacer, expr).to_expr() - - def ensure_expanded_build(build_dir: Path, project_root: Path, expanded: Path) -> Path: """Expand `${TALLYMAN_PROJECT_ROOT}` placeholders into a stable, persistent dir. @@ -117,9 +91,9 @@ def ensure_expanded_build(build_dir: Path, project_root: Path, expanded: Path) - this dir (when the source is snapshotted into `database_tables/`). The read happens at execute time — diff/viewer time, long after load — so deleting the dir first yields a silent zero-row read. - * xorq embeds the build dir path in the expression hash used by - `ParquetSnapshotCache`, so a stable path keeps stat-cache lookups hitting - rather than missing on every fresh tmp path. + * Buckaroo's summary-stat keys include the build directory's path, so a + stable path keeps stat-cache lookups hitting rather than missing on + every fresh tmp path. A sibling `.complete` marker, written last, gates reuse: an expansion interrupted by a crash/OOM leaves a partial dir with no marker and is diff --git a/src/tallyman_xorq/primary_key.py b/src/tallyman_xorq/primary_key.py index f9b0dd6..7708c39 100644 --- a/src/tallyman_xorq/primary_key.py +++ b/src/tallyman_xorq/primary_key.py @@ -34,6 +34,8 @@ from itertools import combinations from pathlib import Path +from tallyman_xorq.row_order import ROW_ORDER + PK_SEARCH_BUDGET_S = 1.0 # Indirection so tests can drive the budget with a fake clock. @@ -196,7 +198,9 @@ def resolve_primary_key( if cached is not None: return cached - cols = set(_entry_columns(project, content_hash)) + # __row_order is unique in every table, so it would win the search for any table without a real key, and row + # positions shift between versions, so a diff keyed on it would be meaningless (ADR-008 D6). + cols = set(_entry_columns(project, content_hash)) - {ROW_ORDER} # Row-preserving revision → inherit the parent's key if it still applies. from tallyman_xorq.result_cache import cache_worthy @@ -285,8 +289,8 @@ def diff_keys( skips the keyed diff (``full_diff(keys=[])``). Both sides share one ``PK_SEARCH_BUDGET_S`` budget. """ - a_cols = set(_entry_columns(project, a_hash)) - b_cols = set(_entry_columns(project, b_hash)) + a_cols = set(_entry_columns(project, a_hash)) - {ROW_ORDER} + b_cols = set(_entry_columns(project, b_hash)) - {ROW_ORDER} deadline = _clock() + PK_SEARCH_BUDGET_S for h in (a_hash, b_hash): pk = resolve_primary_key(project, h, threshold=threshold, max_group=max_group, deadline=deadline) diff --git a/src/tallyman_xorq/result_cache.py b/src/tallyman_xorq/result_cache.py index c76ce2c..15c7212 100644 --- a/src/tallyman_xorq/result_cache.py +++ b/src/tallyman_xorq/result_cache.py @@ -1,41 +1,25 @@ -"""xorq-backed result cache for catalog entries, gated by a structural rubric. - -The entry's frozen build (``xorq_build/``) is the source of truth, and **every -read loads it** — ``ensure_expanded_build`` → ``load_expr(cache_dir=…)`` → the -deep cache-dir rewrite (``portable.rewrite_cache_dirs``). This is the canonical -read of ``docs/system-contract.md`` (#163): a content hash names the fixed -result its build freezes, so reads never re-import ``expr.py`` — recipe -re-execution happens only at minting (build / revise / recalc) and in the -structural-nondeterminism diagnostic below. A missing or unloadable build is a -hard error (ADR D6), never a fallback. - -Whether we keep a materialised copy of the result depends on how expensive the -computation is — caching only pays off when recompute costs far more than -reading a cached parquet: - - * Expensive entries — the expression contains an Aggregate / Join / Sort / - window / UDF — are materialised in xorq's ``ParquetSnapshotCache``. Reads - hit the cache; a deleted cache file self-heals by re-executing the frozen - build, and the healed bytes are verified against the recorded - ``result_digest`` before they are served (ADR D7/D10). - * Cheap entries — a source read plus projections / renames / row-wise scalar - math (the citibike column-reorder/derive case) — materialise nothing (#73). - Recompute ≈ re-reading a columnar source (pushdown), so a stored copy would - burn work and storage for ~no savings; a non-parquet source's parse is - already cached (see ``tallyman_xorq.source_cache``). Reads of a cheap entry - execute the loaded build directly — no per-entry ``result.parquet`` is ever - written, on demand or otherwise. Every consumer (the viewer, paginated - reads, diffs, post-processing) reads ``cached_result_expr``. - -Why ``ParquetSnapshotCache`` over the other xorq caches: parquet storage is -what the DuckDB keyed diff and the Buckaroo viewer stream from and is -inspectable on disk; the *snapshot* strategy keys purely on expression -structure (a catalog entry is an immutable, content-addressed artifact, so its -result must not invalidate just because an upstream file's mtime drifts); and -there's no TTL because entries are permanent history, not expiring scratch. - -(Buckaroo summary stats are cached separately, in each entry's -``.buckaroo_stat_cache`` — always on, orthogonal to this decision.) +"""Reading an entry's result: the canonical read (#163), over tallyman's own materialization. + +The entry's frozen build (``xorq_build/``) is the source of truth, and **every read loads it** — +``ensure_expanded_build`` → ``load_expr``. This is the canonical read of ``docs/system-contract.md``: a content hash +names the fixed result its build freezes, so reads never re-import ``expr.py`` — recipe re-execution happens only at +minting (build / revise / recalc) and in the structural-nondeterminism diagnostic below. A missing or unloadable build +is a hard error (ADR-006 D6), never a fallback. + +There is no xorq cache node in any build (ADR-007 D1). Whether an entry has a file of its own is one recorded fact, +``manifest.cache_worthy``, decided once at build by ``worthiness.classify_expr`` (ADR-008 D4): + + * **Worthy** entries do work that is expensive or that cannot inherit a row order (an aggregate, join, sort, window + function, UDF, union). Tallyman materializes them: ``materialize`` writes + ``compute_cache/result_cache/.parquet`` when the entry is created, and every read is a bare read of that + file. A file that is missing is made again and checked against the recorded digest by ``ensure_materialized`` + before anything reads it (ADR-007 D5). + * **Cheap** entries are row-preserving over one file (a filter, a selection, a computed column). They write nothing; + reading one runs its small frozen plan over files that exist. + +Every consumer (the viewer, paginated reads, diffs, post-processing, chaining a child recipe) reads +``cached_result_expr``. (Buckaroo summary stats are cached separately, in each entry's ``.buckaroo_stat_cache``: always +on, orthogonal to this decision.) """ from __future__ import annotations @@ -43,9 +27,7 @@ import contextvars import functools import logging -import re import sys -import threading import time from pathlib import Path from typing import NamedTuple @@ -54,62 +36,16 @@ # independently of the rest of tallyman's logging (#60), via TALLYMAN_LOG_LEVEL. perf_log = logging.getLogger("tallyman.perf") -# Ops whose presence makes an expression worth caching: they require a -# shuffle / sort / full materialisation rather than a streaming row-wise pass. -_EXPENSIVE_OPS = { - "Aggregate", - "Join", - "JoinChain", - "JoinLink", - "JoinReference", - "Sort", - "SortKey", - "WindowFunction", - "RowNumber", -} - - -def classify_build(build_dir: Path) -> dict: - """Decide whether an entry's result is worth caching, from its build. - - Reads the serialized expression (no execution) and looks for expensive ops - or UDFs. Returns ``{"worthy": bool, "why": str}``. - - A non-parquet *source* read (CSV / JSON) is **not** itself worthy (#73): - the parse is cached at the read by the injected source-cache node (see - ``tallyman_xorq.source_cache``), so an entry earns a ``result_cache`` - snapshot only when it does expensive work — a shuffle / sort / window / UDF - / full materialisation — on top of its source. - - Two implementations of one predicate: this reads the serialized build, - ``source_cache._is_worthy_expr`` walks the live expression to make the same - bake decision at build time. They share ``_EXPENSIVE_OPS`` and must stay in - lockstep — change one, change both. - """ - ops: set[str] = set() - for y in Path(build_dir).glob("*.yaml"): - text = y.read_text() - ops |= set(re.findall(r"op:\s*([A-Za-z_]+)", text)) - - expensive = ops & _EXPENSIVE_OPS - udfs = {o for o in ops if "UDF" in o} - - worthy = bool(expensive or udfs) - why_bits = [] - if expensive: - why_bits.append("ops:" + ",".join(sorted(expensive))) - if udfs: - why_bits.append("udf:" + ",".join(sorted(udfs))) - return { - "worthy": worthy, - "why": "; ".join(why_bits) or "cheap (no Aggregate/Join/Sort/window/UDF)", - } - def cache_worthy(project: str, content_hash: str) -> bool: - from tallyman_core.paths import entry_build_dir + """Whether the entry is materialized, read from its manifest: the verdict recorded at build (ADR-008 D4). - return classify_build(entry_build_dir(project, content_hash))["worthy"] + Nothing re-derives it: the manifest is the record, and ``expr.yaml`` is never parsed to work it out. + """ + from tallyman_core import read_manifest + from tallyman_core.paths import entry_dir + + return bool(read_manifest(entry_dir(project, content_hash)).cache_worthy) # Entries currently being reconstructed by cached_result_expr, on this call @@ -246,44 +182,22 @@ def _recipe_expr(project: str, content_hash: str): pass -def _cached_node_path(baked) -> Path | None: - """On-disk path of the top-level baked result-cache snapshot, or None. - - ``baked`` is ``rewrite_for_build``'s output (the expression carrying the - build's cache nodes). Returns the snapshot path when its top node is the - baked result ``CachedNode``; None when the top isn't a cache — a cheap - expression bakes nothing, or ``classify_build`` (serialized) and - ``_is_worthy_expr`` (live) disagreed. - """ - node = baked.op() - if type(node).__name__ != "CachedNode": - return None - return Path(node.cache.storage.get_path(node.cache.calc_key(node.parent))) - - def load_entry_expr(project: str, content_hash: str): """Load the entry's frozen build as a live expression — the canonical read (#163). - ``ensure_expanded_build`` → ``load_expr(expanded, cache_dir=compute_cache)`` - → ``rewrite_cache_dirs`` (the deep rewrite xorq's shallow load-time one - misses for nested cache nodes, ADR D4). The build binds by value — parents - inlined, sources as content-pinned paths — so the returned expression is the - entry's fixed computation regardless of where alias heads sit today. + ``ensure_expanded_build`` → ``load_expr``. No cache directory is supplied and none is needed (ADR-007 D7): a build + holds no cache node, so nothing in it resolves through one. The build binds by value — sources as content-pinned + ordered copies, a worthy parent as a bare read of its snapshot, a cheap parent's graph inlined — so the returned + expression is the entry's fixed computation regardless of where alias heads sit today. - A missing or unloadable build raises ``BuildError`` naming the entry and the - remedy (ADR D6). There is no recipe fallback: ``expr.py`` binds by name and - re-executing it is how #163's lineage drift happened. + A missing or unloadable build raises ``BuildError`` naming the entry and the remedy (ADR-006 D6). There is no + recipe fallback: ``expr.py`` binds by name and re-executing it is how #163's lineage drift happened. """ from xorq.ibis_yaml.compiler import load_expr - from tallyman_core.paths import ( - compute_cache_dir, - entry_build_dir, - entry_expanded_build_dir, - project_dir, - ) + from tallyman_core.paths import entry_build_dir, entry_expanded_build_dir, project_dir from tallyman_xorq.build import BuildError - from tallyman_xorq.portable import ensure_expanded_build, rewrite_cache_dirs + from tallyman_xorq.portable import ensure_expanded_build build_dir = entry_build_dir(project, content_hash) if not (build_dir / "expr.yaml").is_file(): @@ -293,14 +207,11 @@ def load_entry_expr(project: str, content_hash: str): "path's only source of truth — rebuild the entry (catalog_revise / " "catalog_recalc) to restore it; reads never fall back to expr.py." ) - cache_dir = compute_cache_dir(project) - cache_dir.mkdir(parents=True, exist_ok=True) try: expanded = ensure_expanded_build( build_dir, project_dir(project), entry_expanded_build_dir(project, content_hash) ) - loaded = load_expr(expanded, cache_dir=cache_dir) - return rewrite_cache_dirs(loaded, cache_dir) + return load_expr(expanded) except Exception as exc: raise BuildError( f"entry {content_hash} in {project!r}: loading its frozen build failed " @@ -321,24 +232,22 @@ def _profile_content_token(backend) -> str: return tokenize(toolz.dissoc(backend._profile.as_dict(), "idx")) -def _rebind_to_default_backend(expr): - """Collapse every backend in a loaded build onto the process default backend. - - ``load_expr`` mints fresh backend objects per profile, so two loaded builds — - or a loaded build and a recipe's ``read_project_file`` — span distinct - backend objects and composition raises "Multiple backends found". The - contract allows the collapse because every profile in a tallyman build is - content-identical no-arg ``xorq_datafusion``; that assumption is enforced - here (ADR D3): more than one distinct content profile fails loudly rather - than misbinding a node onto the wrong kind of connection. A raw - ``DatabaseTable`` (bundled data that would need a copy) also fails loudly — - ``replace_sources`` raises unless told to transfer, and tallyman builds must +def rebind_onto(expr, target): + """Collapse every backend in a loaded build onto *target*. + + ``load_expr`` mints fresh backend objects per profile, so two loaded builds — or a loaded build and a recipe's + ``read_project_file`` — span distinct backend objects and composition raises "Multiple backends found". The + contract allows the collapse because every profile in a tallyman build is content-identical no-arg + ``xorq_datafusion``; that assumption is enforced here (ADR-006 D3): more than one distinct content profile fails + loudly rather than misbinding a node onto the wrong kind of connection. A raw ``DatabaseTable`` (bundled data that + would need a copy) also fails loudly — ``replace_sources`` raises unless told to transfer, and tallyman builds must never contain one (in-memory reads are rejected at build). + + The *target* is the process default backend for reads and composition, and the single-partition connection when an + entry is materialized (ADR-009 D1): a loaded build ignores a connection it was never bound to. """ from xorq.common.utils.graph_utils import find_all_sources, replace_sources - from xorq.config import default_backend - target = default_backend() others = [s for s in find_all_sources(expr) if s is not target] if not others: return expr @@ -348,72 +257,38 @@ def _rebind_to_default_backend(expr): raise BuildError( f"loaded build spans {len(tokens)} distinct backend content profiles; " - "rebinding onto the default backend would misbind — every profile in a " - "tallyman build must be content-identical no-arg xorq_datafusion (ADR D3)" + "rebinding onto one backend would misbind — every profile in a " + "tallyman build must be content-identical no-arg xorq_datafusion (ADR-006 D3)" ) return replace_sources({id(s): target for s in others}, expr) -def _assert_recorded_snapshot_key(project: str, content_hash: str, path: Path) -> None: - """ADR D8 tripwire: the read's snapshot derivation must match the build's. - - The build records the baked snapshot's filename (``manifest.snapshot_key``); - the canonical read asserts its own derivation lands on the same file. With - both sides sharing one derivation route the original disagreement mechanism - is gone — a mismatch means an xorq tokenization change or a rewrite drift, - and reading on would serve the wrong file silently. Manifests without the - field (mid-rebuild corpus) are skipped. - """ - from tallyman_core import read_manifest - from tallyman_core.paths import entry_dir - - try: - recorded = read_manifest(entry_dir(project, content_hash)).snapshot_key - except (OSError, ValueError): - return - if recorded is None or path.name == recorded: - return - from tallyman_xorq.build import BuildError +def _rebind_to_default_backend(expr): + """Collapse every backend in a loaded build onto the process default backend (ADR-006 D3).""" + from xorq.config import default_backend - raise BuildError( - f"entry {content_hash} in {project!r}: the read derives snapshot key " - f"{path.name!r} but the build recorded {recorded!r} — the two derivations " - "have diverged (an xorq upgrade or a rewrite drift); rebuild the entry " - "rather than read the wrong snapshot" - ) + return rebind_onto(expr, default_backend()) def baked_snapshot_path(project: str, content_hash: str) -> Path | None: - """Path of the entry's baked result-cache snapshot, or None if it bakes none. - - Derived from the entry's *loaded build* — the same derivation - ``cached_result_expr`` serves and the build recorded — so it names the very - file a cold read returns. None for a cheap entry (bakes nothing), a - worthiness disagreement, or a build written under salt mode (no cache node — - see ``rewrite_for_build``). Raises like ``load_entry_expr`` when the build is - missing (ADR D6). + """Path of the entry's snapshot, or None for a cheap entry, which has none. + + A function of the content hash and the manifest's ``cache_worthy`` (ADR-007 D2), so it loads nothing. """ - return _resolve_result_plan(project, content_hash).path + from tallyman_xorq.materialize import snapshot_path + + return snapshot_path(project, content_hash) if cache_worthy(project, content_hash) else None def snapshot_file_digest(path: Path) -> str: - """SHA-256 of a snapshot parquet file's raw bytes. - - The new ``result_digest`` for worthy entries (ADR - ``ADR-004-result-digest-canonical-ordering.md``): the bake sorts by - ``original_row_order`` before materialising, so the file bytes are - deterministic run-to-run and the file hash is a sound multiset identity. - ~0.3s for a 400 MB file — far cheaper than the retired per-row - ``repr()``+sha256 loop. Cheap entries record no digest (their - recompute is live; there is no snapshot to hash). + """The content digest of a snapshot parquet file: ``arrow-sha256:`` (ADR-009 D2). + + A SHA-256 over the file's ordered Arrow data, read back, so it does not depend on the row-group size, the codec, + the writer's version or how the rows were batched. Cheap entries record no digest (they have no snapshot). """ - import hashlib + from tallyman_xorq.digest import content_digest - h = hashlib.sha256() - with open(path, "rb") as fh: - for chunk in iter(lambda: fh.read(1 << 20), b""): - h.update(chunk) - return h.hexdigest() + return content_digest(Path(path)) def stream_row_count(expr) -> int: @@ -441,20 +316,20 @@ def _recorded_result_digest(project: str, content_hash: str) -> str | None: def verify_result_faithful(project: str, content_hash: str) -> bool | None: - """Whether executing this entry's frozen build reproduces its recorded digest. - - Re-anchored by ADR D7: the snapshot is located through the entry's *own - loaded build* — never today's alias state — so the verdict is about this - entry's bytes, not a sibling's. Returns True when the snapshot file's - SHA-256 matches the recorded ``result_digest``, False on drift, and None - when there is nothing to check (no recorded digest — a cheap entry — or no - snapshot on disk yet; a cold entry heals on its next read, which verifies). + """Whether the entry's snapshot on disk still has its recorded ``result_digest``. + + Returns True when the file's content digest matches, False on drift, and None when there is nothing to check (no + recorded digest, which a cheap entry has, or no snapshot on disk). It reads and never writes (ADR-007 D12): a + snapshot that is missing is checked at the moment it is next made, since every file ``ensure_materialized`` writes + is verified before it is served. """ + from tallyman_xorq.materialize import snapshot_path + recorded = _recorded_result_digest(project, content_hash) if not recorded: return None - snap = _resolve_result_plan(project, content_hash).path - if snap is None or not snap.exists(): + snap = snapshot_path(project, content_hash) + if not snap.exists(): return None return snapshot_file_digest(snap) == recorded @@ -500,19 +375,14 @@ def recipe_is_structurally_nondeterministic(project: str, content_hash: str) -> tell those apart, and an impure UDF's nondeterminism is execution-level anyway (#83), so a UDF entry's drift is attributed execution, never structural. """ - from tallyman_core.paths import entry_build_dir - - # classify_build's why carries "udf:" when the serialized build holds a - # UDF (the same detection cache_worthy uses, #81). Reuse it rather than re-walk. - # Best-effort, like _reconstructed_hash below: classify_build does unguarded - # Path.glob + read_text, so a missing or unreadable build dir raises. This - # predicate runs on the self-heal warning path OUTSIDE its try/except (and, per - # #125, would run at build time too), so a fault here must degrade to "not - # structural" — the conservative execution (#83) attribution — never break the - # read it only annotates. The docstring's "an unresolvable hash returns False" - # contract owns this for every caller. + # The manifest's worthiness reason carries "udf:" when the graph holds a UDF (#81). Best-effort, like + # _reconstructed_hash below: this predicate runs on the self-heal warning path, so a fault here must degrade to + # "not structural" — the conservative execution (#83) attribution — never break the read it only annotates. try: - why = classify_build(entry_build_dir(project, content_hash))["why"] + from tallyman_core import read_manifest + from tallyman_core.paths import entry_dir + + why = read_manifest(entry_dir(project, content_hash)).cache_worthy_why or "" except Exception: return False if "udf:" in why: @@ -524,51 +394,64 @@ def recipe_is_structurally_nondeterministic(project: str, content_hash: str) -> # Called (project, content_hash) after an UNFAITHFUL self-heal, best-effort. -# The companion registers a hook that evicts the entry's Buckaroo session and -# pushes the SSE badge event (ADR D7/D10); processes without an SSE bus (the -# MCP server) still get the durable errors.jsonl record written below. +# The companion registers a hook that forces Buckaroo to reload the entry's grid and pushes the SSE badge event +# (ADR-007 D6); processes without an SSE bus (the MCP server) still get the durable errors.jsonl record written below. UNFAITHFUL_HEAL_HOOKS: list = [] -def _verify_self_heal(project: str, content_hash: str, path: Path) -> None: - """Verify a just-repopulated snapshot against the build-time digest (ADR D7). - - Eviction's load-bearing assumption is that an evicted snapshot recomputes to - what was evicted. A faithful heal is byte-identical (the canonical sort makes - the bake deterministic); a mismatch means this entry's recompute is genuinely - nondeterministic and the heal just manufactured different bytes under the - entry's recorded hash. The read is still served — the bytes are the honest - output of the frozen build — but never silently: - - * a ``tallyman.perf`` UNFAITHFUL warning, attributing the drift as - structural (#88: the recipe's graph hash itself moves) vs execution - (#83: a fixed graph that runs differently); - * a durable ``errors.jsonl`` record (``code="unfaithful_heal"``) — the UI - badge's source, and the ADR D12 pin marking (the entry's bytes are not - regenerable, so eviction machinery must treat its snapshot as retained, - not reclaimable); - * the entry's ``.buckaroo_stat_cache`` is wiped (ADR D10): Buckaroo's - summary stats key on expression structure and stable paths - (buckaroo#955), so stale stats would render beside the fresh rows; - * registered hooks fire (companion: session eviction + SSE). +def _engine_change(project: str, content_hash: str) -> str | None: + """A sentence naming what changed when the engine versions differ from the ones recorded at build (ADR-009 D4).""" + from tallyman_core import read_manifest + from tallyman_core.paths import entry_dir + from tallyman_xorq.materialize import SNAPSHOT_FORMAT_VERSION, engine_versions + + try: + manifest = read_manifest(entry_dir(project, content_hash)) + except (OSError, ValueError): + return None + changes = [] + recorded = manifest.engine_versions or {} + for name, now in engine_versions().items(): + was = recorded.get(name) + if was is not None and was != now: + changes.append(f"{name} {was} -> {now}") + if manifest.snapshot_format is not None and manifest.snapshot_format != SNAPSHOT_FORMAT_VERSION: + changes.append(f"snapshot format {manifest.snapshot_format} -> {SNAPSHOT_FORMAT_VERSION}") + return ", ".join(changes) or None + + +def _verify_self_heal(project: str, content_hash: str, actual: str) -> None: + """Verify a just-repopulated snapshot's content digest against the build-time digest (ADR-007 D5, ADR-006 D7). + + Eviction's load-bearing assumption is that an evicted snapshot recomputes to what was evicted. A faithful heal + has the recorded digest; a mismatch means the entry's recompute changed and the heal just manufactured different + rows under the entry's recorded hash. The read is still served — the bytes are the honest output of the frozen + build — but never silently: + + * a ``tallyman.perf`` UNFAITHFUL warning, attributing the change (ADR-009 D4): the engine (a version recorded + at build differs from today's), the recipe's graph moving (#88), or a fixed graph that runs differently (#83); + * a durable ``errors.jsonl`` record (``code="unfaithful_heal"``) — the UI badge's source, and the pin (the entry's + bytes are not regenerable, so the Cache page's delete leaves its file alone); + * the entry's ``.buckaroo_stat_cache`` is wiped (ADR-006 D10): Buckaroo's summary stats key on expression + structure and stable paths (buckaroo#955), so stale stats would render beside the fresh rows; + * registered hooks fire (companion: a forced reload of the open grid, and the SSE event). """ recorded = _recorded_result_digest(project, content_hash) - if not recorded: - return - try: - actual = snapshot_file_digest(path) - except Exception: - return - if actual == recorded: + if not recorded or actual == recorded: return - kind = ( - "structural (#88) — the recipe bakes a nondeterministic literal that re-derives a different graph hash" - if recipe_is_structurally_nondeterministic(project, content_hash) - else "execution (#83) — a fixed graph that runs differently each execute, or source drift under off" - ) + engine = _engine_change(project, content_hash) + if engine: + kind = ( + f"the engine changed since the entry was built ({engine}), so its result may differ. " + "Rebuild the entry; the recipe is not implicated" + ) + elif recipe_is_structurally_nondeterministic(project, content_hash): + kind = "structural (#88) — the recipe bakes a nondeterministic literal that re-derives a different graph hash" + else: + kind = "execution (#83) — a fixed graph that runs differently each execute, or source drift under off" perf_log.warning( - "cached_result_expr self-heal %s: UNFAITHFUL recompute [%s] — result " - "digest %s != recorded %s; eviction self-healed it to different bytes " + "ensure_materialized self-heal %s: UNFAITHFUL recompute [%s] — result " + "digest %s != recorded %s; eviction self-healed it to different rows " "than were built", content_hash, kind, @@ -587,7 +470,7 @@ def _verify_self_heal(project: str, content_hash: str, path: Path) -> None: project, code="unfaithful_heal", message=( - f"self-heal produced different bytes than were built [{kind}]: " + f"self-heal produced different rows than were built [{kind}]: " f"digest {actual} != recorded {recorded}" ), hash=content_hash, @@ -604,162 +487,100 @@ def _verify_self_heal(project: str, content_hash: str, path: Path) -> None: class _ResultPlan(NamedTuple): """How an entry's result is read, resolved once from its frozen build. - ``loaded`` is the build as ``load_entry_expr`` returned it — per-load backend - objects, exactly the expression shape the build executed, so a heal re-runs - the build's own computation. ``graph`` is the same graph rebound onto the - process default backend (ADR D3) — the composable form chaining and cheap - reads serve. ``path`` is the baked snapshot's location, None when the build - bakes none (cheap entry, worthiness disagreement, or a salt-mode build). + ``loaded`` is the build as ``load_entry_expr`` returned it — per-load backend objects. ``graph`` is the same graph + rebound onto the process default backend (ADR-006 D3) — the composable form a cheap read serves and chaining + inlines. ``reads`` is every file the plan's ``Read`` nodes point at, which ``ensure_materialized`` checks exist + before anything executes (ADR-007 D5). """ - kind: str # "baked" | "recompute" loaded: object graph: object - path: Path | None + reads: tuple[Path, ...] + + +def _read_paths(expr) -> tuple[Path, ...]: + """Every file the expression's ``Read`` nodes point at, in graph order and without repeats.""" + from xorq.common.utils.graph_utils import walk_nodes + from xorq.expr.relations import Read + + seen: dict[Path, None] = {} + for node in walk_nodes(Read, expr): + path = dict(node.read_kwargs).get("hash_path") + if path: + seen[Path(str(path))] = None + return tuple(seen) @functools.lru_cache(maxsize=256) def _resolve_result_plan(project: str, content_hash: str) -> _ResultPlan: - """The expensive, memoisable half of ``cached_result_expr``: load the entry's - frozen build and decide how its result is read. The LRU is sound because its - key finally determines its value — the build is immutable and the plan is a - pure function of it. ``cached_result_expr`` re-checks the snapshot's on-disk - presence on every call, so a snapshot evicted *after* a warm read still - self-heals rather than returning a dangling ``deferred_read_parquet``. + """The expensive, memoisable half of ``cached_result_expr``: load the entry's frozen build and collect what it + reads. The LRU is sound because its key finally determines its value — the build is immutable and the plan is a + pure function of it. Whether each file exists is checked on every call by ``ensure_materialized``, since existence + is the one input that remains mutable. """ t0 = time.monotonic() + loaded = load_entry_expr(project, content_hash) + plan = _ResultPlan(loaded, _rebind_to_default_backend(loaded), _read_paths(loaded)) + perf_log.debug( + "cached_result_expr cold read %s: reads=%d wall_ms=%.1f", + content_hash, + len(plan.reads), + (time.monotonic() - t0) * 1000, + ) + return plan - def _cold(tag): - perf_log.debug( - "cached_result_expr cold read %s: path=%s wall_ms=%.1f", - content_hash, - tag, - (time.monotonic() - t0) * 1000, - ) - loaded = load_entry_expr(project, content_hash) - graph = _rebind_to_default_backend(loaded) - path = _cached_node_path(loaded) - if path is None: - # No baked result cache in the build: a cheap entry, a - # classify_build/_is_worthy_expr disagreement, or a salt-mode build - # (rewrite_for_build bakes no cache under salt). Execute the frozen - # graph directly. - _cold("cheap-recompute") - return _ResultPlan("recompute", loaded, graph, None) - _assert_recorded_snapshot_key(project, content_hash, path) - _cold("baked-read") - return _ResultPlan("baked", loaded, graph, path) - - -# Per-(project, content_hash) locks serialise the snapshot heal in cached_result_expr -# so two concurrent cold reads of the same entry don't both execute the shared baked op -# (#79). Keyed locks under a master lock; the registry grows with distinct entries read -# (bounded by catalog size) and stale locks are harmless. -_HEAL_LOCKS: dict[tuple[str, str], threading.Lock] = {} -_HEAL_LOCKS_GUARD = threading.Lock() - - -def _heal_lock(project: str, content_hash: str) -> threading.Lock: - key = (project, content_hash) - with _HEAL_LOCKS_GUARD: - lock = _HEAL_LOCKS.get(key) - if lock is None: - lock = threading.Lock() - _HEAL_LOCKS[key] = lock - return lock +@functools.lru_cache(maxsize=1024) +def _snapshot_read(project: str, content_hash: str): + """One bare read of the entry's snapshot, memoised for the life of the process (ADR-007 D2). + + A read has one table name, so repeated reads stop piling up tables in the shared backend. xorq still registers a + deferred read's table on every execute, so the footer is opened once per query, which the snapshot format keeps + small. + """ + from xorq.expr.api import deferred_read_parquet + + from tallyman_xorq.materialize import snapshot_path + + return deferred_read_parquet(str(snapshot_path(project, content_hash))) def cached_result_expr(project: str, content_hash: str): """The entry's result as a single-backend expression on the default backend. - Both paths below serve the entry's *frozen build* (``load_entry_expr``), - root on the default backend, and materialise no entry ``result.parquet``: - - * Cheap entry — the loaded build's graph, rebound onto the default backend - (ADR D3). Recompute on read is ~free (pushdown over content-pinned - columnar sources; a non-parquet parse is cached at the read), and the - graph is frozen, so the bytes are the entry's recorded result — never - today's alias heads. - * Expensive entry — read its baked ``result_cache`` snapshot directly as a - single ``deferred_read_parquet`` on the default backend. The snapshot was - materialised when the entry built (``rewrite_for_build``'s top-level - result cache); reading it skips re-running the DAG and keeps the cache - node's own storage connection out of the caller's expression. Self-heals: - an evicted snapshot is re-executed once *from the frozen build*, verified - against the recorded digest (ADR D7), then read. - - For composition that must stay self-healing — chaining a child recipe off - this entry — use ``entry_graph_expr`` instead: it returns the graph with the - cache node still in place, so the child's build carries its parents' healing - knowledge (ADR D4). - - Bounded LRU so the cache can't grow without limit; entries are - content-addressed and builds immutable, so an evicted hash reloads an - identical expression. - - A salt-mode build carries no cache nodes (path-only snapshot keys would - collide across salted entries — see ``rewrite_for_build``), so a salted entry - falls through to the recompute path — single-backend, content-honest, no - parquet. - - The build load is memoised in ``_resolve_result_plan``; the snapshot's - on-disk presence is re-checked HERE on every call, so an expensive entry - whose snapshot is evicted *after* a warm read self-heals on the next read - instead of handing back a stale ``deferred_read_parquet`` (which would - ``ValueError: At least one path is required`` at execute time). The - perf-namespace cold-read tag (#87) is emitted from the memoised half, so it - still fires once per genuine load. - """ - from xorq.expr.api import deferred_read_parquet + Every file the read needs is made to exist first (``ensure_materialized``, ADR-007 D5), so nothing executes over a + missing file. - plan = _resolve_result_plan(project, content_hash) - if plan.kind == "recompute": - return plan.graph - path = plan.path - if not path.exists(): - # Single-flight the heal per (project, content_hash). _resolve_result_plan is - # lru-cached, so concurrent cold readers share one loaded op; without this lock - # both miss the evicted snapshot and both execute it, and the loser raises — - # DataFusion "Already borrowed" on the shared in-process op, or FileNotFoundError - # on xorq ParquetStorage's fixed .parquet.tmp across processes — surfacing - # as a 500 from api_data and, swallowed by ensure_session, an empty grid (#79). - with _heal_lock(project, content_hash): - if not path.exists(): # a peer thread may have healed it while we waited - try: - # Evicted snapshot: re-execute the frozen build once, then read. - # plan.loaded is the exact expression shape the build executed, - # so a faithful heal lands byte-identical parquet. - plan.loaded.count().execute() - except (FileNotFoundError, ValueError): - # A peer *process* (MCP build vs companion heal sharing compute_cache) - # can win the shared-tmp rename out from under us. If the snapshot - # landed, read it; otherwise the failure is real, so re-raise. - if not path.exists(): - raise - perf_log.debug("cached_result_expr self-heal %s: evicted-self-heal", content_hash) - _verify_self_heal(project, content_hash, path) - return deferred_read_parquet(str(path)) - - -def entry_graph_expr(project: str, content_hash: str): - """The entry's frozen graph — cache nodes included — on the default backend. - - The chaining primitive behind ``tracked_expr_from_alias`` / - ``pinned_expr_from_alias`` (ADR D4). Where ``cached_result_expr`` serves a - worthy entry as a bare read of its snapshot file (fast, but a file reference - with no regeneration knowledge), this returns the loaded build itself: the - entry's ``CachedNode`` stays in the graph, so a child recipe built over it - freezes its parents' healing knowledge into its own build — a cold load of - the child regenerates ancestor snapshots through ordinary cache mechanics, - with no pre-heal choreography. Rebound onto the process default backend so - it composes with ``read_project_file`` and other entries as one backend. + * Worthy entry: ONE bare read of its snapshot (``_snapshot_read``, memoised), served without loading the entry's + build when the file exists. Composing it into a child recipe makes the child's identity a function of the + parent's, since the path carries the parent's content hash (ADR-007 D3). + * Cheap entry: the loaded build's graph, rebound onto the default backend (ADR-006 D3). Its small plan re-runs on + every read over files that exist, and the graph is frozen, so the rows are the entry's recorded result and + never today's alias heads. """ + from tallyman_xorq.materialize import _ensure + + if _ensure(project, content_hash): + return _snapshot_read(project, content_hash) return _resolve_result_plan(project, content_hash).graph -# Existing callers clear the result-expr memo via cached_result_expr.cache_clear() -# (companion app, conftest, cache tests); the memo now lives on the plan resolver, -# so re-expose its cache controls on the public name. -cached_result_expr.cache_clear = _resolve_result_plan.cache_clear +def preload_plan(project: str, content_hash: str) -> None: + """Load an entry's frozen build into the in-process memo. Writes nothing (ADR-007 D12). + + The startup warm-up uses it, so the first page request of a cheap entry does not pay for the load. It does not call + ``ensure_materialized``, and it needs no file to exist: loading a build does not open the files it reads. + """ + _resolve_result_plan(project, content_hash) + + +def _clear_read_memos() -> None: + _resolve_result_plan.cache_clear() + _snapshot_read.cache_clear() + + +# Existing callers clear the result memo via cached_result_expr.cache_clear() (companion app, conftest, reset, recalc, +# tests); the memos now live on the plan resolver and the snapshot read, so re-expose their cache controls on the public +# name. +cached_result_expr.cache_clear = _clear_read_memos cached_result_expr.cache_info = _resolve_result_plan.cache_info diff --git a/src/tallyman_xorq/row_order.py b/src/tallyman_xorq/row_order.py new file mode 100644 index 0000000..f20d911 --- /dev/null +++ b/src/tallyman_xorq/row_order.py @@ -0,0 +1,287 @@ +"""``__row_order``: the natural order of a file, and the sorts that keep every query deterministic (ADR-008). + +Every file tallyman writes ends in an int64 column named ``__row_order`` holding ``0..N-1`` in the file's physical row +order. A page of any entry is ``ORDER BY __row_order`` (or the user's keys, then ``__row_order``), so it is the same +page in any process and any cache state. This module holds the rules that keep that true while recipes are written by +someone else: + +- a cheap entry has no file of its own, so it inherits the column and must keep it (D3); +- a recipe may read the column and copy it under another name, but not assign to it (D6); +- every sort in a recipe gets the natural order, then the remaining columns, as its last keys, so the sort is total + wherever the recipe put it (D10), and a sort that is not the last step decides the order that is written (D11). +""" + +from __future__ import annotations + +ROW_ORDER = "__row_order" +# ibis's name for the right side's copy of a column both sides of a join carry. +ROW_ORDER_RIGHT = "__row_order_right" + + +class RowOrderError(RuntimeError): + """A recipe breaks the row-order contract. ``build_and_persist`` reports it as a ``BuildError``.""" + + +def _sortable(dtype) -> bool: + """Whether a column can serve as a sort key. + + Nested / geospatial types can't be sort keys in datafusion; leaving them out of the tie-breakers is safe, since + any remaining ties are between rows identical on every sortable column, and identical rows write identical bytes. + """ + for pred in ("is_array", "is_map", "is_struct", "is_json", "is_geospatial"): + if getattr(dtype, pred, None) and getattr(dtype, pred)(): + return False + return True + + +def _constant_names(rel) -> set[str]: + """Columns the relation defines as scalars (projected literals/constants). + + A constant column adds no ordering information (every row ties on it), and ibis dereferences a sort key straight + through to the defining value, so ``ORDER BY `` reaches datafusion's planner, which rejects a bare + literal there (it reads a numeric literal as a column ordinal). Skipping them is both necessary and free. + """ + values = getattr(rel, "values", None) + if not values: + return set() + out = set() + for name, v in values.items(): + shape = getattr(v, "shape", None) + if shape is not None and shape.is_scalar(): + out.add(name) + return out + + +def tie_break_order(names, keyed: set[str], schema, rel) -> list[str]: + """The columns to append to a sort: ``__row_order`` first, then every other sortable, non-constant column.""" + skip = keyed | _constant_names(rel) + remaining = [n for n in names if n not in skip and _sortable(schema[n])] + remaining.sort(key=lambda n: n != ROW_ORDER) # stable: the natural order first, schema order after + return remaining + + +def _keyed_names(keys) -> set[str]: + import xorq.vendor.ibis.expr.operations as ops + + return {k.expr.name for k in keys if isinstance(k.expr, ops.Field)} + + +def _extend_every_sort(expr): + """Append the tie-break to every ``Sort`` in the graph (ADR-008 D10).""" + import xorq.vendor.ibis.expr.operations as ops + from xorq.common.utils.graph_utils import replace_nodes, walk_nodes + + if not walk_nodes(ops.Sort, expr): + return expr + + def replacer(node, kwargs): + node = node.__recreate__(kwargs) if kwargs else node + if not isinstance(node, ops.Sort): + return node + parent = node.parent + remaining = tie_break_order(parent.schema.names, _keyed_names(node.keys), parent.schema, parent) + if not remaining: + return node + extra = tuple(ops.SortKey(ops.Field(parent, n)) for n in remaining) + return ops.Sort(parent, tuple(node.keys) + extra) + + return replace_nodes(replacer, expr).to_expr() + + +def _map_key_through(step, name: str) -> str: + """Follow one sort key (a column name) up through one order-keeping step, or say why it did not survive.""" + import xorq.vendor.ibis.expr.operations as ops + + if isinstance(step, ops.Project): + outputs = [ + out + for out, v in step.values.items() + if isinstance(v, ops.Field) and v.name == name and v.rel == step.parent + ] + if name in outputs: + return name + if outputs: + return outputs[0] # a rename is followed + raise RowOrderError( + f"the sort key {name!r} was " + + ("overwritten" if name in step.values else "dropped") + + " by a later step" + ) + if isinstance(step, ops.DropColumns): + if name in step.columns_to_drop: + raise RowOrderError(f"the sort key {name!r} was dropped by a later step") + return name + if isinstance(step, ops.FillNull): + replacements = step.replacements + if not isinstance(replacements, dict) or name in replacements: + raise RowOrderError(f"the sort key {name!r} was overwritten by a later fill_null") + return name + return name # Filter, Limit, DropNull: the columns are the parent's + + +def _hoisted_keys(expr): + """The keys the author wrote in the nearest sort reachable from the top through order-keeping steps, as + ``(output column, ascending, nulls_first)`` for the top-level sort. + + None when no sort is reachable (an aggregate or a join sits in the way, and the order of rows is gone there). + Raises ``RowOrderError`` when a key did not survive as an unchanged output column (ADR-008 D11). Only the keys + the author wrote count: the tie-break appended to a sort (``__row_order`` and the other columns) is added again at + the top, and a later select is free to drop those columns. + """ + import xorq.vendor.ibis.expr.operations as ops + + top = expr.op() + chain = [] + node = top + while not isinstance(node, ops.Sort): + if not isinstance(node, (ops.Filter, ops.Limit, ops.DropNull, ops.Project, ops.DropColumns, ops.FillNull)): + return None + chain.append(node) + node = node.parent + hoisted = [] + for key in node.keys: + if not isinstance(key.expr, ops.Field): + raise RowOrderError( + "an order_by that is not the last step must sort by plain columns so its order can be kept, and " + f"this one sorts by an expression ({key.expr.name}). Sort as the last step, or add the expression as " + "a column first" + ) + name = key.expr.name + try: + for step in reversed(chain): + name = _map_key_through(step, name) + except RowOrderError as exc: + raise RowOrderError( + f"{exc}, so the order you asked for cannot be kept. Keep the column in every later select, or " + "sort as the last step of the recipe" + ) from exc + hoisted.append((name, key.ascending, key.nulls_first)) + return hoisted + + +def canonical_sorted(expr): + """Impose a deterministic total order on a worthy entry before it is materialized. + + Key priority: the author's own sort keys stay primary (the served row order is part of what they asked for), then + ``__row_order``, then the remaining sortable columns in schema order. Every ``Sort`` in the recipe is extended + that way in place (D10). When the recipe's last step is not a sort, the nearest sort below it that the order of + rows survives to is hoisted, so the top-level sort leads with its keys (D11); with none, the top-level sort is the + tie-break alone. + """ + import xorq.vendor.ibis.expr.operations as ops + + lead = _hoisted_keys(expr) if not isinstance(expr.op(), ops.Sort) else None + expr = _extend_every_sort(expr) + node = expr.op() + if isinstance(node, ops.Sort): + return expr + lead_keys = [ops.SortKey(ops.Field(node, name), asc, nulls) for name, asc, nulls in (lead or [])] + schema = expr.schema() + names = tie_break_order(schema.names, {name for name, _, _ in (lead or [])}, schema, node) + keys = tuple(lead_keys) + tuple(ops.SortKey(ops.Field(node, n)) for n in names) + if not keys: + return expr # nothing sortable: the digest stays best-effort for this entry + return ops.Sort(node, keys).to_expr() + + +def assert_not_assigned(expr) -> None: + """A recipe may read ``__row_order`` and copy it under another name, but not assign to it (ADR-008 D6). + + Arbitrary values could hold ties or gaps, and the contract depends on ``0..N-1`` with neither. Passing the column + through unchanged is not an assignment. + """ + import xorq.vendor.ibis.expr.operations as ops + from xorq.common.utils.graph_utils import walk_nodes + from xorq.vendor.ibis.expr.operations.core import Node + + def _is_passthrough(value) -> bool: + return isinstance(value, ops.Field) and value.name == ROW_ORDER + + for node in walk_nodes((Node,), expr): + assigned = False + if isinstance(node, ops.Project): + value = node.values.get(ROW_ORDER) + assigned = value is not None and not _is_passthrough(value) + elif isinstance(node, ops.Aggregate): + value = node.groups.get(ROW_ORDER) + assigned = (value is not None and not _is_passthrough(value)) or ROW_ORDER in node.metrics + if assigned: + raise RowOrderError( + f"the recipe assigns to {ROW_ORDER!r}, which is reserved: it holds each row's position in the file " + "(0..N-1) and paging depends on that. To change the order of rows, sort them (order_by) and the " + f"column is renumbered to match. To keep a copy, give it another name, e.g. " + f"t.mutate({ROW_ORDER}_v1=t[{ROW_ORDER!r}])" + ) + + +def require_on_cheap(expr, *, reading: str) -> None: + """A cheap entry has no file of its own, so it must carry ``__row_order`` from what it reads (ADR-008 D3).""" + columns = list(expr.columns) + if ROW_ORDER in columns: + return + fix = ", ".join(repr(c) for c in [*columns, ROW_ORDER]) + raise RowOrderError( + f"this entry reads {reading} and keeps its rows, so it must keep the {ROW_ORDER!r} column: pages of the " + f"entry are ordered by it, which is what keeps paging repeatable. Add it to the select, e.g. " + f"t.select({fix})" + ) + + +def move_last(expr): + """Put ``__row_order`` in the last position, at the top of the expression only. + + A computed column added after it would otherwise push it into the middle of the table. + """ + columns = list(expr.columns) + if columns[-1] == ROW_ORDER: + return expr + return expr.select(*[c for c in columns if c != ROW_ORDER], ROW_ORDER) + + +def assert_joinable(expr) -> None: + """A join chain whose first side and two or more right-hand sides all carry ``__row_order`` collides (ADR-008 D6). + + Every entry carries the column, so a join of two entries leaves the right side's copy behind under ibis's + collision name, and the second join in the same chain needs that name too. Whether ibis then fails while building + the expression, while compiling it, or not at all depends on what is stacked on top, so the check is explicit and + the author is told what to write. + """ + import xorq.vendor.ibis.expr.operations as ops + from xorq.common.utils.graph_utils import walk_nodes + + for chain in walk_nodes(ops.JoinChain, expr): + with_column = [link for link in chain.rest if ROW_ORDER in link.table.schema.names] + if ROW_ORDER in chain.first.schema.names and len(with_column) >= 2: + raise _collision_error() + + +def _collision_error() -> RowOrderError: + return RowOrderError( + f"joining more than two entries in one recipe: every entry carries {ROW_ORDER!r}, so the second join collides " + f"on {ROW_ORDER_RIGHT!r}. Drop the column from the right-hand inputs, e.g. " + f"a.join(b.drop({ROW_ORDER!r}), key).join(c.drop({ROW_ORDER!r}), key)" + ) + + +def translate_collision(exc: Exception) -> RowOrderError | None: + """The instruction for ibis's raw name-collision error on ``__row_order_right``, or None for any other error. + + A join of two entries leaves the right side's copy behind under ibis's collision name, and a second join in the + same recipe collides with it. The author never wrote that name, so the raw message would mean nothing to them. + """ + if ROW_ORDER_RIGHT not in str(exc): + return None + return _collision_error() + + +def page(expr, *, offset: int, limit: int, sort=()): + """One page of an entry: ``ORDER BY , __row_order`` then ``LIMIT/OFFSET`` (ADR-008 D5). + + With no user sort the page is ``ORDER BY __row_order``. With one, the user's keys come first and ``__row_order`` + is the last key, which breaks every tie, so the same request returns the same rows in any process and any cache + state. An expression that does not carry the column (a diff, which drops it) is paged as given. + """ + keys = [*sort, ROW_ORDER] if ROW_ORDER in expr.columns else list(sort) + if keys: + expr = expr.order_by(keys) + return expr.limit(limit, offset=offset) diff --git a/src/tallyman_xorq/source_cache.py b/src/tallyman_xorq/source_cache.py index 9816fe9..aa493e5 100644 --- a/src/tallyman_xorq/source_cache.py +++ b/src/tallyman_xorq/source_cache.py @@ -1,245 +1,98 @@ -"""Rewrite-then-build source caching for catalog entries. - -Before a build, tallyman rewrites the submitted expression: every non-parquet -*file* read (``read_csv`` / ``read_json``) gets a ``.cache()`` node injected -immediately downstream, so the CSV→parquet parse is materialised once and -shared — the snapshot strategy keys on the read's path only, so every entry -that reads the same source hits the same cached parquet. ``read_parquet`` / -``read_delta`` are exempt: already columnar, re-reading them costs about the -same as reading a cached copy. - -An in-memory read (``read_in_memory`` or an ``ibis.memtable``) is a hard error. -It signals the author dropped to pandas (e.g. ``pd.read_csv`` then -``read_in_memory``) instead of a native deferred reader; the bytes would not -round-trip through the portable build, and what actually wants caching is the -native parse. The error steers the author back to ``deferred_read_csv``. - -The author never writes ``.cache()``. ``build_and_persist`` builds the -*rewritten* expression, so the entry's ``content_hash`` and ``xorq_build/`` -always carry the cache node while ``expr.py`` keeps the LLM's literal source — -tallyman does not depend on the submitted form (the LLM doesn't reliably track -it). The injected cache's ``base_path`` is supplied at load time by xorq's -``load_expr(cache_dir=...)`` (``compiler.replace_base_path`` rewrites every -``CachedNode`` storage to that dir), which tallyman points at the project's -``compute_cache`` — so the source-read parquet lands under the per-project, -reset-reconciled compute cache, content-addressed and self-healing. -""" +"""The rewrite step between a recipe and its build (ADR-007 D1, ADR-008). -from __future__ import annotations +Before a build, tallyman looks at the submitted expression, rejects what cannot become a sound entry, and adds the one +thing a materialized entry needs. No xorq cache node is created or kept: tallyman writes its own result files +(``tallyman_xorq.materialize``), so nothing in a build points at xorq's cache, and xorq stays the expression, build, +load, hashing and execution layer. -# File reads cheap enough to re-read rather than cache: columnar formats whose -# scan is already a pushdown read. Everything else (read_csv, read_json) parses -# text and is worth materialising once. -_EXEMPT_READS = frozenset({"read_parquet", "read_delta"}) +The rewrite rejects: +- **In-memory reads** (``read_in_memory`` or an ``ibis.memtable``). They signal that the author dropped to pandas + instead of reading a file through ``read_project_file`` or ``tallyman_read_csv``; the bytes would not round-trip + through the portable build. +- **A recipe that already contains a cache node.** A node with default storage would write under ``~/.cache/xorq``. +- **A recipe that assigns to ``__row_order``**, and **a cheap entry that drops it** (ADR-008 D3, D6). -def _should_cache_read(method_name: str) -> bool: - """True if a deferred file read should get a source-cache node injected. +And it adds: - Membership-based, not format-special-cased: every non-exempt reader - (``read_csv``, ``read_json``) is cached; columnar formats are exempt. Kept a - named predicate so the read_csv/read_json parity stays unit-testable — this - xorq has no ``deferred_read_json``, so the read_json path has no end-to-end - producer to build a fixture from. - """ - return method_name not in _EXEMPT_READS +- for a **worthy** entry, the canonical sort (ADR-006 D5, ADR-008 D10 and D11): a deterministic total order, so the + snapshot is the same on every rebuild and the writer numbers the rows in the order the author asked for; +- for a **cheap** entry, nothing but moving ``__row_order`` to the last column. +Whether an entry is cheap or worthy is decided once, by ``worthiness.classify_expr`` on the expression the author +wrote, and recorded in the manifest. This module never re-derives it from the serialized build. +""" -def _sortable(dtype) -> bool: - """Whether a column can serve as a canonical-order sort key. +from __future__ import annotations - Nested / geospatial types can't be sort keys in datafusion; leaving them out - of the tie-breakers is safe — any remaining ties are between rows identical - on every sortable column, and identical prefixes write identical bytes. - """ - for pred in ("is_array", "is_map", "is_struct", "is_json", "is_geospatial"): - if getattr(dtype, pred, None) and getattr(dtype, pred)(): - return False - return True +from tallyman_xorq.row_order import ( + canonical_sorted as _canonical_sorted, # noqa: F401 (the evidence scripts import it here) +) -def _constant_names(rel) -> set[str]: - """Columns the relation defines as scalars (projected literals/constants). +class InMemoryReadError(RuntimeError): + """The expression reads in-memory data instead of a deferred file read.""" - A constant column adds no ordering information — every row ties on it — and - ibis dereferences a sort key straight through to the defining value, so - ``ORDER BY `` reaches datafusion's planner, which rejects a bare - literal there ("invalid digit found in string": it reads a numeric literal - as a column ordinal). Skipping them is both necessary and free. - """ - values = getattr(rel, "values", None) - if not values: - return set() - out = set() - for name, v in values.items(): - shape = getattr(v, "shape", None) - if shape is not None and shape.is_scalar(): - out.add(name) - return out - - -def _tie_break_order(names, keyed: set[str], schema, rel) -> list[str]: - """Sortable, non-constant columns not already keyed, ``original_row_order`` first. - - ``original_row_order`` is tallyman's canonical total order for CSV-sourced - entries (``tallyman_read_csv``); leading with it keeps a baked CSV result in - file order rather than lexicographic-by-first-column order. - """ - skip = keyed | _constant_names(rel) - remaining = [n for n in names if n not in skip and _sortable(schema[n])] - remaining.sort(key=lambda n: n != "original_row_order") # stable: row order first, schema order after - return remaining - - -def _canonical_sorted(expr): - """Impose a deterministic total order before the result-cache bake. - - ADR D5 (amended, plans/ADR-006-read-path-loads-builds.md): datafusion's parallel - scan/aggregation reorders rows run-to-run above its 1 MiB repartition - threshold, so an unsorted bake produces different bytes on every heal and - ``result_digest`` stops naming the entry's result. Sorting by every column - makes the bytes deterministic: rows still tied after all sortable columns are - byte-identical, so any permutation among them writes the same file. - - Key priority: the author's own top-level ``order_by`` keys stay primary (the - served row order is part of what they asked for), then ``original_row_order`` - (the canonical CSV file order), then the remaining sortable columns in schema - order. An author sort is extended in place rather than wrapped, so the graph - carries one Sort node. - """ - import xorq.vendor.ibis.expr.operations as ops - node = expr.op() - if isinstance(node, ops.Sort): - parent = node.parent - keyed = {k.expr.name for k in node.keys if isinstance(k.expr, ops.Field)} - remaining = _tie_break_order(parent.schema.names, keyed, parent.schema, parent) - if not remaining: - return expr - extra = tuple(ops.SortKey(ops.Field(parent, n)) for n in remaining) - return ops.Sort(parent, tuple(node.keys) + extra).to_expr() - schema = expr.schema() - names = _tie_break_order(schema.names, set(), schema, node) - if not names: - return expr # nothing sortable — the digest stays best-effort for this entry - return expr.order_by(names) +class CacheNodeError(RuntimeError): + """The recipe contains a xorq cache node.""" _IN_MEMORY_MSG = ( "expression reads in-memory data (read_in_memory / ibis.memtable). This " "usually means the source was loaded into pandas (e.g. pd.read_csv) and " - "handed to xorq in memory, instead of a native deferred reader. Read the " + "handed to xorq in memory, instead of a native reader. Read the " "file with tallyman_read_csv(abs_path, schema=...) for CSVs or " - "xo.deferred_read_parquet(abs_path) for parquet, so the source round-trips " - "through the portable build and is cached at the read." + "read_project_file(rel_path) for parquet, so the source round-trips " + "through the portable build and is ingested with a stable __row_order." ) +_CACHE_NODE_MSG = ( + "the recipe calls .cache(), which puts a xorq cache node in the build. Tallyman writes its own result files " + "(entries that do expensive work are materialized when they are created), and a cache node with default storage " + "would write under ~/.cache/xorq. Remove the .cache() call." +) -class InMemoryReadError(RuntimeError): - """The expression reads in-memory data instead of a deferred file read.""" +def rewrite_for_build(expr, project: str, *, verdict=None, reading: str | None = None): + """Rewrite the submitted expression before build. -def _is_worthy_expr(expr) -> bool: - """True if the expression does expensive work worth materialising. + In order: - Mirrors ``result_cache.classify_build`` (which decides the same thing from - the serialized build) on the *live* expression, so the bake decision below - and the read-time worthiness check stay in lockstep: an Aggregate / Join / - Sort / window / UDF anywhere in the graph makes the result worth caching. + 1. **Reject in-memory reads** — raise :class:`InMemoryReadError`. + 2. **Reject cache nodes** — raise :class:`CacheNodeError` (ADR-007 D1). + 3. **Reject assignment to ``__row_order``**, and, for a cheap entry, a select that drops it — raise + :class:`tallyman_xorq.row_order.RowOrderError` (ADR-008 D3, D6). ``reading`` names what the entry reads, for + the message. + 4. A worthy entry gets the **canonical sort**; a cheap entry gets ``__row_order`` moved last. - UDFs are matched by ancestry, not leaf name. A ``make_pandas_udf`` node is - classed after the user's function (e.g. ``plusone``), so its live leaf name - carries no "UDF" — but its MRO holds the base op name (``ScalarUDF`` / - ``AggUDF`` / …) that the serialized YAML emits and ``classify_build`` greps. - Testing the leaf name only missed scalar UDFs and broke the lockstep (#81). - """ - from xorq.common.utils.graph_utils import walk_nodes - from xorq.vendor.ibis.expr.operations.core import Node - - from tallyman_xorq.result_cache import _EXPENSIVE_OPS - - for node in walk_nodes((Node,), expr): - if type(node).__name__ in _EXPENSIVE_OPS: - return True - if any("UDF" in base.__name__ for base in type(node).__mro__): - return True - return False - - -def rewrite_for_build(expr, project: str): - """Rewrite the submitted expression before build (#73). - - Three rewrites, in order: - - 1. **Reject in-memory reads** (always). ``read_in_memory`` / an - ``ibis.memtable`` signals the author dropped to pandas instead of a - native deferred reader — raise :class:`InMemoryReadError`. - 2. **Cache each non-parquet source read.** A ``.cache()`` after every - ``read_csv`` / ``read_json`` materialises the parse once, shared across - entries (path-only snapshot key). ``read_parquet`` / ``read_delta`` are - exempt — re-reading a columnar source is already cheap. - 3. **Bake a top-level result cache when the expression is expensive.** The - cache node becomes part of the durable recipe, so *every* loader — the - Buckaroo viewer, diffs, ``tracked_expr_from_alias`` chaining — reads the cached - result instead of re-running the whole DAG (the bug #71 papered over at - read time). A cheap expression bakes nothing and recomputes ~free. - - All cache nodes resolve their storage at load time to the per-project - compute cache (xorq's ``load_expr(cache_dir=…)`` rewrites every node's base - path uniformly), so there is no separate ``result_cache/`` dir — source - parses and baked results co-locate there, content-addressed and - reset-reconciled. - - Cache injection is skipped under ``salt`` source-identity mode: its - path-only snapshot keys would collide across salted entries with identical - paths but different content. In-memory rejection still applies. + ``verdict`` is the entry's ``worthiness.Verdict`` (computed here when the caller has not already). """ import xorq.vendor.ibis.expr.operations as ops - from xorq.caching import ParquetSnapshotCache - from xorq.common.utils.graph_utils import replace_nodes, walk_nodes - from xorq.expr.relations import Read + from xorq.common.utils.graph_utils import walk_nodes + from xorq.expr.relations import CachedNode, Read - from tallyman_core.paths import compute_cache_dir - from tallyman_xorq import source_identity as si - from tallyman_xorq.backend import connect + from tallyman_xorq import row_order + from tallyman_xorq.worthiness import classify_expr reads = walk_nodes(Read, expr) if walk_nodes(ops.InMemoryTable, expr) or any(r.method_name == "read_in_memory" for r in reads): raise InMemoryReadError(_IN_MEMORY_MSG) - - if si.mode() == "salt": - return expr - - base = str(compute_cache_dir(project)) - - # 2. Source-read caches. One cache (one connection) shared by every read; - # the snapshot key is path-only so distinct reads get distinct keys. The - # `wrapped` memo breaks re-descent recursion: replace_nodes re-applies the - # replacer to a fresh CachedNode's `parent` (graph_utils process_node), the - # same Read — without the memo it would wrap it again, forever. - targets = {r for r in reads if _should_cache_read(r.method_name)} - if targets: - src_cache = ParquetSnapshotCache.from_kwargs(source=connect(), base_path=base) - wrapped: dict = {} - - def replacer(node, kwargs): - if isinstance(node, Read) and node in targets: - if node in wrapped: - return node - cached = node.to_expr().cache(cache=src_cache).op() - wrapped[node] = cached - return cached - return node.__recreate__(kwargs) if kwargs else node - - expr = replace_nodes(replacer, expr).to_expr() - - # 3. Bake the result cache for an expensive expression, canonically ordered - # first so the snapshot's bytes — and therefore result_digest — are - # deterministic across the build and every later heal (ADR D5, amended). - # A distinct relative_path keeps baked results inspectable apart from - # source parses under the same compute-cache root. - if _is_worthy_expr(expr): - result_cache = ParquetSnapshotCache.from_kwargs(source=connect(), base_path=base, relative_path="result_cache") - expr = _canonical_sorted(expr).cache(cache=result_cache) - - return expr + if walk_nodes(CachedNode, expr): + raise CacheNodeError(_CACHE_NODE_MSG) + + verdict = verdict or classify_expr(expr) + row_order.assert_not_assigned(expr) + row_order.assert_joinable(expr) + if verdict.worthy: + try: + return row_order.canonical_sorted(expr) + except row_order.RowOrderError: + raise + except Exception as exc: + translated = row_order.translate_collision(exc) + if translated is not None: + raise translated from exc + raise + row_order.require_on_cheap(expr, reading=reading or "a file") + return row_order.move_last(expr) diff --git a/src/tallyman_xorq/staleness.py b/src/tallyman_xorq/staleness.py index 897cc77..b9812ff 100644 --- a/src/tallyman_xorq/staleness.py +++ b/src/tallyman_xorq/staleness.py @@ -116,20 +116,21 @@ def entry_staleness(project: str, content_hash: str) -> StaleVerdict: def verify_sweep(project: str) -> dict: - """Opt-in corpus verification (ADR D7): do baked snapshots still match their - recorded ``result_digest``? - - For every entry that recorded a digest, ``verify_result_faithful`` locates - the snapshot through the entry's own frozen build and compares file hashes. - Returns ``{"results": {hash: bool|None}, "unfaithful": [...], "errors": - {hash: message}}`` — ``None`` means nothing to check yet (snapshot not on - disk; the next read heals and verifies), ``errors`` carries entries whose - build failed to load (the D6 hard error, reported per-entry so one broken - entry can't abort a corpus sweep). + """Opt-in corpus verification (ADR-006 D7): do materialized snapshots still match their recorded ``result_digest``? + + For every entry that recorded a digest, ``verify_result_faithful`` compares the content digest of the snapshot on + disk with the recorded one. It READS AND NEVER WRITES (ADR-007 D5, D12): a snapshot that is missing stays missing, + because a sweep that rewrote every deleted file would undo the Cache page's delete, and every file + ``ensure_materialized`` writes is verified before it is served, so an absent file is checked at the moment it next + exists. Returns ``{"results": {hash: bool|None}, "unfaithful": [...], "absent": [...], "errors": {hash: message}}``: + ``None`` in ``results`` means nothing to check (the snapshot is absent), ``absent`` lists those hashes, and + ``errors`` carries entries whose check failed (reported per-entry so one broken entry can't abort a corpus sweep). """ + from tallyman_xorq.materialize import snapshot_path from tallyman_xorq.result_cache import verify_result_faithful results: dict[str, bool | None] = {} + absent: list[str] = [] errors: dict[str, str] = {} for entry in list_entries(project): if not entry.get("result_digest"): @@ -137,11 +138,14 @@ def verify_sweep(project: str) -> dict: h = entry["content_hash"] try: results[h] = verify_result_faithful(project, h) + if not snapshot_path(project, h).exists(): + absent.append(h) except Exception as exc: errors[h] = str(exc) return { "results": results, "unfaithful": sorted(h for h, ok in results.items() if ok is False), + "absent": sorted(absent), "errors": errors, } diff --git a/src/tallyman_xorq/worthiness.py b/src/tallyman_xorq/worthiness.py new file mode 100644 index 0000000..4d8d33d --- /dev/null +++ b/src/tallyman_xorq/worthiness.py @@ -0,0 +1,86 @@ +"""Cheap or worthy: the one place that decides (ADR-008 D4). + +A **cheap** entry has no file of its own: its small plan re-runs on every read, and it pages by the ``__row_order`` +of the one file it reads. So cheap has to mean something strong, that the plan is row-preserving over exactly one file +and that ``__row_order`` is still present and unique at the top of it. Everything else is **worthy**, which means it +is materialized: a snapshot is written when the entry is created and pages read that file. + +The test is an allow-list, so an operation nobody has thought about yet costs a copy (safe) instead of unstable +paging (unsafe). It has three parts, all decided on the live expression by the class of each node: + +- every relation operation is on ``ROW_PRESERVING_RELATIONS``; +- the plan reads exactly one file; +- no value operation multiplies rows (``Unnest``), depends on the order rows arrive in (``WindowFunction``, which also + covers ``row_number`` and ``lag``) or is not pure (``Impure``: ``random()`` and ``uuid()``; ``now()`` and + ``today()``, which xorq's ibis classes as constants, by name; and any UDF, matched by its base class). + +The verdict is computed once, when the entry is built, and recorded in the manifest. Nothing reads ``expr.yaml`` to +work it out again: a regex over that file cannot hold an allow-list, since ``op:`` in it also matches types and +literals that are not operations. +""" + +from __future__ import annotations + +from typing import NamedTuple + + +class Verdict(NamedTuple): + worthy: bool + why: str + + +# Value operations that break the row-order contract of a cheap entry, by base class. +_NEVER_CHEAP_VALUES = ("Unnest", "WindowFunction", "Impure") + +# xorq's ibis classes these two as constants, not as impure, so they are matched by name. +_NEVER_CHEAP_BY_NAME = frozenset({"TimestampNow", "DateNow"}) + + +def _cheap_relation_types(): + """The relation operations that keep each output row tied to exactly one input row.""" + import xorq.vendor.ibis.expr.operations as ops + from xorq.expr.relations import Read + + return (Read, ops.Filter, ops.Project, ops.DropColumns, ops.DropNull, ops.FillNull) + + +def classify_expr(expr) -> Verdict: + """Cheap or worthy, from the live author expression, with the reason in a short string. + + ``why`` names what made the entry worthy (``ops:Aggregate,Join``, ``reads:2``, ``values:Unnest``, + ``udf:plusone``), or says it is cheap. It is recorded in the manifest as ``cache_worthy_why``. + """ + import xorq.vendor.ibis.expr.operations as ops + from xorq.common.utils.graph_utils import walk_nodes + from xorq.expr.relations import Read + from xorq.vendor.ibis.expr.operations.core import Node + + cheap_relations = _cheap_relation_types() + never_cheap_values = tuple(getattr(ops, name) for name in _NEVER_CHEAP_VALUES) + + nodes = list(walk_nodes((Node,), expr)) + relations = [n for n in nodes if isinstance(n, ops.Relation)] + bits: list[str] = [] + + other = sorted({type(n).__name__ for n in relations if not isinstance(n, cheap_relations)}) + if other: + bits.append("ops:" + ",".join(other)) + reads = {n for n in relations if isinstance(n, Read)} + if len(reads) != 1: + bits.append(f"reads:{len(reads)}") + values = sorted( + { + type(n).__name__ + for n in nodes + if isinstance(n, never_cheap_values) or type(n).__name__ in _NEVER_CHEAP_BY_NAME + } + ) + if values: + bits.append("values:" + ",".join(values)) + udfs = sorted({type(n).__name__ for n in nodes if any("UDF" in base.__name__ for base in type(n).__mro__)}) + if udfs: + bits.append("udf:" + ",".join(udfs)) + + if bits: + return Verdict(True, "; ".join(bits)) + return Verdict(False, "cheap (row-preserving over one file)") From 74372b1558b27bf9d92c1beec4991bf8d54849ce Mon Sep 17 00:00:00 2001 From: Paddy Mullen Date: Mon, 21 Sep 2026 00:33:40 -0400 Subject: [PATCH 008/111] feat(companion): hand Buckaroo files that exist, page by __row_order, pin what cannot be re-made ADR-007 D6 and D12, ADR-008 D5 and D8, ADR-009 D6. - load_session runs ensure_materialized first, then posts /load_expr with a session id derived from the project and the hash. A worthy entry is handed a view build of its snapshot, a cheap entry its own build. Tallyman keeps no session record; a klass reload posts /reload_expr per entry and treats a 404 as not open; an unfaithful heal forces a reload of the open grid. - Every /load_expr names __row_order as the row-order column. - /api/data pages by ORDER BY __row_order. - The startup warm-up loads cheap entries' builds and writes nothing. - The Cache page lists every snapshot file (a file whose entry is gone is an orphan row), marks a snapshot pinned when it cannot be re-created faithfully, and refuses to delete it with a reason. - A diff drops __row_order from both sides. The MCP tool descriptions explain the column and the build errors around it. Co-Authored-By: Claude Sonnet 5 --- src/tallyman_companion/app.py | 223 +++++---- src/tallyman_companion/buckaroo_lifecycle.py | 496 +++++++++---------- src/tallyman_companion/diff.py | 11 +- src/tallyman_mcp/server.py | 43 +- 4 files changed, 391 insertions(+), 382 deletions(-) diff --git a/src/tallyman_companion/app.py b/src/tallyman_companion/app.py index 0027a65..b87ceb7 100644 --- a/src/tallyman_companion/app.py +++ b/src/tallyman_companion/app.py @@ -49,7 +49,8 @@ read_prompts, ) from tallyman_xorq.primary_key import PrimaryKeySearchTimeout, diff_keys -from tallyman_xorq.result_cache import baked_snapshot_path, cached_result_expr +from tallyman_xorq.result_cache import cached_result_expr +from tallyman_xorq.row_order import page as row_order_page log = logging.getLogger("tallyman.companion") @@ -206,15 +207,15 @@ def _compute_disk_usage(project: str) -> dict: def _snapshot_cache_path(project: str, content_hash: str): - """Resolve the xorq snapshot-cache path for one entry, without materialising. + """Resolve the snapshot path for one entry, without materialising. Returns ``(applicable, path | None, note)``. Delegates to - ``result_cache.baked_snapshot_path``, which returns None for an entry that - bakes no snapshot — a cheap expression, salt identity mode, or a worthiness - disagreement — and the snapshot path otherwise (the very file a cold read - returns). It never calls ``.execute()``, so a cold expensive entry reports an - absent (0 B) snapshot rather than being forced to materialise just because - someone opened the metadata tab. + ``result_cache.baked_snapshot_path``, which returns None for a cheap entry, + which has no snapshot, and the snapshot path otherwise (a function of the + content hash and the manifest's ``cache_worthy``). It loads and writes + nothing, so a worthy entry whose file was deleted reports an absent (0 B) + snapshot rather than being forced to materialise just because someone opened + the metadata tab. """ try: from tallyman_xorq.result_cache import baked_snapshot_path # noqa: PLC0415 @@ -223,7 +224,7 @@ def _snapshot_cache_path(project: str, content_hash: str): except Exception as exc: # noqa: BLE001 — best-effort sizing, never 500 the tab return True, None, f"snapshot path unresolved: {type(exc).__name__}" if path is None: - return False, None, "cheap entry — recomputed from the build, no snapshot copy" + return False, None, "cheap entry — a small plan over files that exist, no snapshot" return True, path, "" @@ -265,14 +266,14 @@ def add(key: str, label: str, path: Path, kind: str, reclaimable: bool, detail: components.append( { "key": "snapshot_cache", - "label": "xorq snapshot cache", + "label": "snapshot", "kind": "cache", "reclaimable": True, "applicable": applicable, "exists": snap_exists, "bytes": snap_size, "formatted": _fmt_bytes(snap_size), - "detail": note or "expensive entry — a cached copy in compute_cache/, recomputed on a miss", + "detail": note or "materialized entry — its result in compute_cache/, made again and verified on a miss", } ) @@ -444,12 +445,12 @@ def _build_compare_expr(project: str, a_hash: str, b_hash: str, keys: tuple[str, def _invalidate_reset_caches(project: str | None = None) -> None: """Drop the companion's process-global result/compare caches after a reset. - A reset prunes baked snapshots (``catalog_state.reset_to`` → - ``prune_compute_cache``). ``_build_compare_expr`` serializes those snapshot - paths into a build with no per-call ``exists()`` recheck, so a warmed diff - pair would otherwise serve a build over a parquet the prune deleted — - buckaroo reads zero files → empty compare grid (#80). ``cached_result_expr`` - self-heals on every read so clearing its memo is hygiene, not load-bearing, + A reset changes which entries exist (``catalog_state.reset_to`` retires and + restores entry dirs, and leaves ``compute_cache/`` alone, ADR-007 D14). + ``_build_compare_expr`` serializes snapshot paths into a build with no + per-call ``exists()`` recheck, so a warmed diff pair is dropped rather than + served over entries the reset retired (#80). ``cached_result_expr`` makes its + files exist on every read, so clearing its memo is hygiene, not load-bearing, but we clear it for parity. Blunt global clear is correct and cheap: entries are content-addressed, so the next call rebuilds an identical expression. @@ -671,13 +672,11 @@ async def publish(event: dict): for q in list(subscribers): await q.put(event) - # ADR D7/D10: when a self-heal fails verification (result_cache writes the - # durable errors.jsonl record and wipes the entry's stat cache itself), the - # companion additionally evicts the entry's Buckaroo session — an open - # session's row cache would show the pre-heal rows beside fresh stats — and - # pushes the SSE event the UI badges from. Registered on startup (the hook - # needs the running loop to publish from the sync heal path) and removed on - # shutdown so test-created apps don't pile up dead hooks. + # ADR-006 D7/D10: when a self-heal fails verification (result_cache writes the durable errors.jsonl record and + # wipes the entry's stat cache itself), the companion additionally forces Buckaroo to re-run its pipeline for the + # entry's grid, since an open session holds stats computed from the old rows (ADR-007 D6), and pushes the SSE event + # the UI badges from. Registered on startup (the hook needs the running loop to publish from the sync heal path) + # and removed on shutdown so test-created apps don't pile up dead hooks. @app.on_event("startup") async def _register_unfaithful_heal_hook(): from tallyman_xorq.result_cache import UNFAITHFUL_HEAL_HOOKS # noqa: PLC0415 @@ -686,7 +685,7 @@ async def _register_unfaithful_heal_hook(): def _on_unfaithful_heal(project: str, content_hash: str) -> None: if buckaroo: - buckaroo.evict_session(content_hash) + buckaroo.force_reload_session(project, content_hash) asyncio.run_coroutine_threadsafe( publish({"kind": "unfaithful_heal", "hash": content_hash, "project": project}), loop, @@ -792,8 +791,13 @@ async def _log_revision(): @app.on_event("startup") async def _warm_expr_cache(): + """Load the frozen builds of cheap entries into the in-process memo, so the first page request does not pay + for the load. It never writes a file (ADR-007 D12): a worthy entry is served from its snapshot without loading + its build, and a snapshot the user deleted stays deleted until something is about to read it.""" from starlette.concurrency import run_in_threadpool # noqa: PLC0415 + from tallyman_xorq.result_cache import cache_worthy, preload_plan # noqa: PLC0415 + project = _current_project() if project is None: return @@ -823,7 +827,9 @@ def _warm(): ) return try: - cached_result_expr(project, h) + if cache_worthy(project, h): + continue + preload_plan(project, h) warmed += 1 log.debug( "expr cache warmed %s (%.0fms, %.0fms elapsed)", @@ -912,22 +918,18 @@ def api_data(project: str, content_hash: str, offset: int = 0, limit: int = 200) if not entry_build_dir(project, content_hash).is_dir(): raise HTTPException(404, "no entry") - # #90: serve the page off the entry's expression. cached_result_expr - # already hands back the result as a live single-backend expression, so - # the window pushes down (expensive → windowed read of the baked snapshot, - # cheap → limit pushed through to the source read) and nothing is written. - # The old path called ensure_result, materialising the entry's entire - # result to disk to serve one page. total is the manifest's row_count - # (recorded at build; api_entry_detail reads it the same way), so no file - # is needed to populate it. The manifest is written after the build dir - # exists (build.py), so guard the read: a half-built or pruned entry still - # serves its page off the expression with a best-effort total of 0 rather - # than 500ing on a missing manifest — cached_result_expr needs none. + # A page is a function of (content_hash, sort, offset, limit) (ADR-008 D1, D5): ``ORDER BY __row_order``, or + # the user's keys and then ``__row_order``, so the same request returns the same rows in any process and any + # cache state. cached_result_expr hands back the result as a live single-backend expression over files that + # exist (ensure_materialized ran first), so the window pushes down and nothing is written here. total is the + # manifest's row_count (recorded at build; api_entry_detail reads it the same way). The manifest is written + # after the build dir exists (build.py), so guard the read: a half-built or pruned entry still serves its page + # with a best-effort total of 0 rather than 500ing on a missing manifest. manifest_path = entry_dir(project, content_hash) / ENTRY_MANIFEST_FILENAME total = 0 if manifest_path.exists(): total = json.loads(manifest_path.read_text()).get("row_count") or 0 - df = cached_result_expr(project, content_hash).limit(limit, offset=offset).execute() + df = row_order_page(cached_result_expr(project, content_hash), offset=offset, limit=limit).execute() return { "data": json.loads(df.to_json(orient="records")), "offset": offset, @@ -1501,70 +1503,81 @@ def api_result_cache(project: str): project = _validate_project(project) import datetime # noqa: PLC0415 - from tallyman_core.paths import entries_dir as _entries_dir # noqa: PLC0415 + from tallyman_xorq.materialize import pinned_reason, snapshots_dir # noqa: PLC0415 - root = _entries_dir(project) + # The page lists the snapshot files that exist on disk right now: the rows the delete button can actually + # evict (ADR-007 D2, D12). An entry's snapshot is a function of its hash, so nothing is derived or loaded. A + # file whose entry is not in the catalog (a reset retired the entry and left its file, ADR-007 D14) gets a row + # of its own, marked orphan, since nothing else lists it and the user has to be able to delete it. entries = [] - if root.is_dir(): - for entry in root.iterdir(): - if not entry.is_dir(): - continue - try: - m = read_manifest(entry) - except Exception: - continue - content_hash = entry.name - # The page lists the baked snapshots that exist on disk right now — - # the rows the delete button can actually evict. cache_bytes (#87) is - # a cheap pre-filter: None for a cheap entry (bakes nothing), so skip - # it without deriving a path. For an expensive entry, resolve the - # snapshot and skip it when the file is gone — a delete unlinks the - # snapshot but leaves cache_bytes set, so keying the listing on - # cache_bytes alone showed a phantom row (stale size, freed bytes - # still in the total, a delete button that 404s) until the entry was - # next viewed and re-baked. Size from the live file so the figure - # tracks disk. baked_snapshot_path re-imports the recipe, but only - # for the (few) expensive entries the pre-filter lets through, and - # this admin page isn't on a hot path. - if m.cache_bytes is None: - continue + root = snapshots_dir(project) + for snap in sorted(root.glob("*.parquet")) if root.is_dir() else []: + content_hash = snap.stem + try: + size = snap.stat().st_size + except OSError: + continue + try: + m = read_manifest(entry_dir(project, content_hash)) + except Exception: + m = None + if m is None: + mtime = datetime.datetime.fromtimestamp(snap.stat().st_mtime, tz=datetime.timezone.utc) try: - snap = baked_snapshot_path(project, content_hash) - if snap is None or not snap.exists(): - continue - size = snap.stat().st_size + import pyarrow.parquet as pq # noqa: PLC0415 + + row_count = int(pq.ParquetFile(snap).metadata.num_rows) except Exception: - continue - # created_at is an ISO-8601 UTC string; render it in the table's - # existing "%Y-%m-%d %H:%M:%S" shape (CachePage renders `created` - # verbatim) so the Created column is unchanged. - try: - created = datetime.datetime.fromisoformat(m.created_at).strftime("%Y-%m-%d %H:%M:%S") - except (ValueError, TypeError): - created = m.created_at or "" - alias = alias_for_hash(project, content_hash) - info = version_of_hash(project, content_hash) - version = info[1] if info else None - is_current = alias is not None - if not is_current and info is not None: - alias = info[0] - is_current = False + row_count = 0 entries.append( { "hash": content_hash, "size": size, "size_formatted": _fmt_bytes(size), - # row_count is int|None on the manifest, but CacheEntry - # types it number and CachePage calls .toLocaleString() - # with no null guard — coerce so a null can't reach JS. - "row_count": int(m.row_count or 0), - "created": created, - "alias": alias, - "version": version, - "is_current": is_current, - "prompt": m.prompt, + "row_count": row_count, + "created": mtime.strftime("%Y-%m-%d %H:%M:%S"), + "alias": None, + "version": None, + "is_current": False, + "prompt": None, + "orphan": True, + "pinned": False, + "pinned_reason": None, } ) + continue + # created_at is an ISO-8601 UTC string; render it in the table's existing "%Y-%m-%d %H:%M:%S" shape + # (CachePage renders `created` verbatim) so the Created column is unchanged. + try: + created = datetime.datetime.fromisoformat(m.created_at).strftime("%Y-%m-%d %H:%M:%S") + except (ValueError, TypeError): + created = m.created_at or "" + alias = alias_for_hash(project, content_hash) + info = version_of_hash(project, content_hash) + version = info[1] if info else None + is_current = alias is not None + if not is_current and info is not None: + alias = info[0] + is_current = False + reason = pinned_reason(project, content_hash) + entries.append( + { + "hash": content_hash, + "size": size, + "size_formatted": _fmt_bytes(size), + # row_count is int|None on the manifest, but CacheEntry types it number and CachePage calls + # .toLocaleString() with no null guard — coerce so a null can't reach JS. + "row_count": int(m.row_count or 0), + "created": created, + "alias": alias, + "version": version, + "is_current": is_current, + "prompt": m.prompt, + "orphan": False, + "pinned": reason is not None, + "pinned_reason": reason, + } + ) entries.sort(key=lambda e: -e["size"]) total = sum(e["size"] for e in entries) return { @@ -1582,24 +1595,24 @@ async def api_delete_result_cache(project: str, content_hash: str): # Reject a non-hex hash up front: a 400 reads truer than the # "already evicted" 404 below (see _require_hash). _require_hash(content_hash) - # Evict the baked .cache() snapshot — the single materialised copy. None - # for a cheap entry (nothing to delete) or one already evicted; reading - # the entry's viewer page afterward self-heals (re-bakes) it. - p = baked_snapshot_path(project, content_hash) - if p is None or not p.exists(): - raise HTTPException(404, "no baked snapshot for this entry") + from tallyman_xorq.materialize import pinned_reason, snapshot_path # noqa: PLC0415 + + # Delete the entry's snapshot, or an orphan file's: the single materialized copy. A file is deleted only by an + # explicit user action (ADR-007 D12), and reading the entry afterward makes it again and verifies it. A file + # that cannot be made again faithfully is pinned (ADR-009 D6) and this leaves it alone, with the reason. + p = snapshot_path(project, content_hash) + if not p.exists(): + raise HTTPException(404, "no snapshot for this entry") + reason = pinned_reason(project, content_hash) + if reason: + raise HTTPException(409, f"not deleted: {reason}") try: p.unlink() except OSError as exc: raise HTTPException(500, str(exc)) - # Defense-in-depth, not load-bearing for correctness: since ee0a90a, - # cached_result_expr is a thin non-memoised wrapper that re-checks - # path.exists() and self-heals on every read, so the next viewer read - # (api_data) / ensure_session heal re-bakes the evicted snapshot even with a - # warm memo left in place (proven by - # test_cached_result_expr_self_heals_after_warm_then_evict). We still clear - # it because it's cheap for an admin delete: content-addressed entries just - # re-reconstruct (and the evicted one re-bakes) on the next read. + # The memoised read of the snapshot outlives the file, but every read goes through ensure_materialized first, + # so the next read of the entry makes it again. Clearing the memo is hygiene: content-addressed entries just + # reload an identical plan. cached_result_expr.cache_clear() return {"ok": True, "hash": content_hash} diff --git a/src/tallyman_companion/buckaroo_lifecycle.py b/src/tallyman_companion/buckaroo_lifecycle.py index df1f044..d4dae6a 100644 --- a/src/tallyman_companion/buckaroo_lifecycle.py +++ b/src/tallyman_companion/buckaroo_lifecycle.py @@ -4,7 +4,10 @@ sort/filter/search) and is what beat 2 of the talk's storyboard relies on. Buckaroo runs as its own Tornado server on a separate port; the companion mounts the React embed (``static/buckaroo-embed.js``) into the entry-detail -page and pre-warms a WS session per catalog entry. +page and opens a WS session per catalog entry. + +Buckaroo is a displayer (ADR-007, governing rule): it runs queries only for summary stats, sorting and paging. +Tallyman runs an entry's computation to completion first, and hands Buckaroo something that already exists. Lifecycle: @@ -16,21 +19,17 @@ `BUCKAROO_PORT=` so we recover the bound port (useful when port=0). 3. Poll `/health` until Buckaroo reports ready (~200ms typical). -4. When the entry-detail route is hit for the first time on a content - hash, POST `/load_expr` with the entry's `xorq_build/` dir and store - the returned `session` in `catalog/buckaroo_sessions.json` for later - retrieval. Buckaroo serves the entry via the xorq backend with - push-down sort/search (paging over a materialised parquet is the - `/load` flow, which we no longer use). +4. When the entry-detail route is hit on a content hash, make sure every file the entry reads exists + (``ensure_materialized``), then POST `/load_expr` with a build dir and a session id derived from the project and + the hash. A worthy entry is handed a *view build*, a build whose whole graph is one read of its snapshot; a cheap + entry is handed its own build, a small plan over files that exist. 5. On shutdown (uvicorn lifespan or `atexit`), close the subprocess's stdin and wait briefly; if it doesn't go, SIGTERM. -The session map is persisted so that across companion restarts we don't -re-call /load redundantly on the same hash. Buckaroo itself is fresh on -each start, though, so the cached session_ids only act as a *naming* -convention — Buckaroo's own state is rebuilt on first hit either way. -This means: a `tallyman serve` restart re-loads parquets lazily as entries -are viewed; that's correct. +Tallyman keeps no record of Buckaroo's sessions (ADR-007 D6). The session id is a function of the project and the +content hash, so it is never stale: a repeat POST is a no-op in Buckaroo while it holds the session (same id, same +build dir, none of the config-bearing fields), and re-creates it if Buckaroo has dropped it, as Buckaroo does after an +idle hour. """ from __future__ import annotations @@ -42,6 +41,7 @@ import socket import subprocess import sys +import tempfile import threading import time from pathlib import Path @@ -52,22 +52,56 @@ entry_build_dir, entry_expanded_build_dir, entry_stat_cache_dir, + entry_view_build_dir, ) from tallyman_core.manifest import read_manifest from tallyman_core.paths import artifacts_dir, entry_manifest_path, project_dir +from tallyman_xorq.row_order import ROW_ORDER + +log = logging.getLogger("tallyman.buckaroo") + +# One view build is written per entry directory at a time (a per-path lock, like ``ensure_expanded_build``'s). +_view_locks_guard = threading.Lock() +_view_locks: dict[str, threading.Lock] = {} -def _entry_exists(project: str, content_hash: str) -> bool: - """True if the project's catalog has a built entry for *content_hash*. +def ensure_view_build(project: str, content_hash: str) -> Path: + """The stable per-entry directory holding the *view build* of a worthy entry's snapshot (ADR-007 D6). - The build dir is the entry's existence proof now (there is no on-demand - ``result.parquet``). Used by ``_load_session_file`` to prune stale session - entries whose project / hash combination no longer maps to a build on disk. + A view build is a xorq build whose whole graph is one step, "read this parquet file". Buckaroo's stat-cache keys + include the build directory's path, so the directory is stable, written once and reused. A sibling marker records + the snapshot path the build was made for, so a project that moved (a clone at another path) regenerates it instead + of pointing Buckaroo at a path that is gone. The snapshot must exist (``ensure_materialized`` has run). """ - return entry_build_dir(project, content_hash).is_dir() + from xorq.expr.api import deferred_read_parquet + from xorq.ibis_yaml.compiler import build_expr + from tallyman_xorq.materialize import snapshot_path -log = logging.getLogger("tallyman.buckaroo") + dest = entry_view_build_dir(project, content_hash) + snap = str(snapshot_path(project, content_hash)) + marker = dest.with_name(dest.name + ".complete") + + def _fresh() -> bool: + try: + return (dest / "expr.yaml").is_file() and marker.read_text() == snap + except OSError: + return False + + if _fresh(): + return dest + with _view_locks_guard: + lock = _view_locks.setdefault(str(dest), threading.Lock()) + with lock: + if _fresh(): + return dest + dest.parent.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(prefix=".xorq_view_build.", dir=dest.parent) as tmp: + built = Path(build_expr(deferred_read_parquet(snap), builds_dir=Path(tmp))) + shutil.rmtree(dest, ignore_errors=True) + shutil.move(str(built), str(dest)) + marker.write_text(snap) + return dest def _port_in_use(port: int) -> bool: @@ -97,19 +131,11 @@ class BuckarooUnavailable(RuntimeError): class BuckarooManager: - """Owns a Buckaroo server subprocess and a multi-project session cache. - - Sessions are keyed by content hash (globally unique by construction), so - one subprocess can serve sessions backed by xorq builds from any project. - The owning project travels per-call via ``ensure_session(hash, project)`` - and is recorded alongside the session id so a restart-reload knows which - project's parquet to find. - - Session state persists in a single global file at - ``~/.tallyman/buckaroo_sessions.json``. The per-project location used - by V0.5 (``/artifacts/catalog/buckaroo_sessions.json``) is gone. - Will be replaced wholesale by a session-enumeration endpoint when - buckaroo-data/buckaroo#860 lands; until then the file is the bookkeeping. + """Owns a Buckaroo server subprocess and opens sessions on it. + + A session's id is a function of the project and the entry's content hash (``session_id_for``), so one subprocess + serves sessions for entries of any project and tallyman needs no record of which are open: Buckaroo is the only + process that knows, and it answers a repeat ``/load_expr`` for a session it holds without redoing the work. """ def __init__( @@ -132,13 +158,10 @@ def __init__( self.companion_base_url = companion_base_url.rstrip("/") if companion_base_url else None self.proc: subprocess.Popen | None = None self._client = httpx.Client(timeout=5.0) - # Schema: {: {"session_id": str, "project": str}}. - self._sessions: dict[str, dict] = {} # Diff-compare session_ids (``diff--``) Buckaroo has loaded this - # lifetime. Reset on a Buckaroo restart, same as ``_sessions`` — these - # sessions live only in the subprocess's RAM. + # lifetime. Reset on a Buckaroo restart: these sessions live only in the + # subprocess's RAM. (The live diff still posts an unmaterialized join, ADR-007 D10, #188.) self._loaded_diff_sessions: set[str] = set() - self._session_lock = threading.Lock() self._buckaroo_started_at: float | None = None # Tmp dirs we created by expanding ${TALLYMAN_PROJECT_ROOT} placeholders # before POSTing /load_expr. Buckaroo holds the loaded xorq expression @@ -151,7 +174,15 @@ def __init__( # ``_restart_cooldown`` seconds. self._last_restart_attempt: float = 0.0 self._restart_cooldown: float = 30.0 - self._load_session_file() + + @staticmethod + def session_id_for(project: str, content_hash: str) -> str: + """The Buckaroo session id of an entry: ``entry--``. + + A function of the two and nothing else (ADR-007 D6), so it is never stale and there is nothing to remember. The + project is in it so that one project's session is never served to another on a hash collision (#172). + """ + return f"entry-{project}-{content_hash}" # ------------------------------------------------------------------ # lifecycle @@ -308,150 +339,82 @@ def mark_diff_session_loaded(self, session_id: str) -> None: self._loaded_diff_sessions.add(session_id) def _reset_session_bookkeeping_if_restarted(self, started_at) -> None: - """Drop in-RAM session bookkeeping when a fresh Buckaroo is detected. - - Buckaroo's ``/load_expr`` sessions — entry (``_sessions``) and - diff-compare (``_loaded_diff_sessions``) alike — live only in the - subprocess's memory, so a restart (a new ``started_at`` from ``/health``) - invalidates every session_id we've handed out. Clearing both forces a - re-POST on next access; that reload is cheap because the on-disk - stat/result cache survives the restart. Keeping a stale entry instead - would hand a client a dead session. + """Drop in-RAM diff-session bookkeeping when a fresh Buckaroo is detected. + + The diff-compare sessions live only in the subprocess's memory, so a restart (a new ``started_at`` from + ``/health``) invalidates every one we've handed out. Clearing forces a re-POST on next access; that reload is + cheap because the on-disk stat cache survives the restart. Entry sessions need no such record (ADR-007 D6): + their ids are derived and every open re-posts. """ if started_at == self._buckaroo_started_at: return - if self._sessions: - log.info( - "buckaroo restart detected; invalidating %d cached sessions", - len(self._sessions), - ) - self._sessions = {} self._loaded_diff_sessions.clear() self._buckaroo_started_at = started_at - self._persist_sessions() # ------------------------------------------------------------------ - # session map (persisted to disk) + # klass reload and forced reload (no session record needed) # ------------------------------------------------------------------ - def _session_file_path(self) -> Path: - from tallyman_core.paths import buckaroo_sessions_path - - return buckaroo_sessions_path() - - def _load_session_file(self) -> None: - """Load the global session map, dropping entries whose parquet has - disappeared since the last persist. - - Schema: ``{"sessions": {hash: {session_id, project}}, "buckaroo_started_at": float}``. - Each entry's ``project`` is treated as a hint — if that project has - no entry for the hash, the session is stale and gets dropped. - """ - p = self._session_file_path() - if not p.exists(): - return - try: - data = json.loads(p.read_text()) - except json.JSONDecodeError: - return - self._buckaroo_started_at = data.get("buckaroo_started_at") - sessions = data.get("sessions", {}) - # Defensive: silently drop entries that aren't the new shape. - kept: dict[str, dict] = {} - for h, info in sessions.items(): - if not isinstance(info, dict): - continue - project = info.get("project") - if not project: - continue - if not _entry_exists(project, h): - continue - kept[h] = {"session_id": info["session_id"], "project": project} - self._sessions = kept - - def _persist_sessions(self) -> None: - p = self._session_file_path() - p.parent.mkdir(parents=True, exist_ok=True) - p.write_text( - json.dumps( - { - "buckaroo_started_at": self._buckaroo_started_at, - "sessions": self._sessions, - }, - indent=2, - sort_keys=True, - ) - ) - def reload_project_sessions(self, project: str) -> int: - """Hot-reload klasses for all live sessions belonging to *project*. + """Hot-reload klasses for every open grid of *project*. - Calls POST /reload_expr/ (buckaroo 0.14.9+) on each - cached session for *project*. The session stays alive and its - analysis/post-processing klasses are updated in place — no eviction - or page-load round-trip to /load_expr is needed. + A klass is a project-authored stat, post-processing or display class. Buckaroo 0.15.6 has no route that lists + its sessions and tallyman keeps no record of them, so this posts ``/reload_expr/`` (buckaroo + 0.14.9+) for each entry of the project and treats the 404 (or 400) Buckaroo answers for an id it does not hold + as "not open" (ADR-007 D6). That is one request per entry per klass change. The session stays alive and its + klasses are updated in place — no page-load round-trip to /load_expr is needed. - After a successful reload the on-disk parquet stat cache for that - entry is cleared so the next widget request recomputes all stats - (including any newly added ones) from scratch. Without this, a stat - added after the session was first loaded would be absent from the - cache and silently omitted from the display. + After a successful reload the on-disk stat cache for that entry is cleared so the next widget request + recomputes all stats (including any newly added ones) from scratch. Without this, a stat added after the + session was first loaded would be absent from the cache and silently omitted from the display. - Returns the number of sessions reloaded. Falls back to 0 (with a - warning) if buckaroo isn't running or a reload call fails. + Returns the number of sessions reloaded. Falls back to 0 (with a warning) if buckaroo isn't running or a + reload call fails. """ if not self.is_running or self.bound_port is None: return 0 - with self._session_lock: - targets = {h: info["session_id"] for h, info in self._sessions.items() if info.get("project") == project} + from tallyman_xorq.build import list_entries + reloaded = 0 - to_evict: list[str] = [] - for content_hash, session_id in targets.items(): + for entry in list_entries(project): + content_hash = entry["content_hash"] + session_id = self.session_id_for(project, content_hash) try: - resp = self._client.post( - f"{self.base_url}/reload_expr/{session_id}", - timeout=5.0, - ) + resp = self._client.post(f"{self.base_url}/reload_expr/{session_id}", timeout=5.0) if resp.status_code in (404, 400): - # Session is gone or is no longer an xorq session — evict - # so the next ensure_session call re-creates it cleanly. - log.warning( - "buckaroo session %s (hash %s) is stale (%d); evicting", - session_id, - content_hash, - resp.status_code, - ) - to_evict.append(content_hash) - continue + continue # Buckaroo does not hold this session (never opened, or idle-evicted) resp.raise_for_status() self._clear_stat_cache(project, content_hash) reloaded += 1 log.info("reloaded klasses for session %s (hash %s)", session_id, content_hash) except httpx.HTTPError as exc: log.warning("buckaroo /reload_expr failed for session %s: %s", session_id, exc) - if to_evict: - with self._session_lock: - for h in to_evict: - self._sessions.pop(h, None) - self._persist_sessions() return reloaded - def evict_session(self, content_hash: str) -> bool: - """Drop the cached session for one entry so its next view reloads fresh. + def force_reload_session(self, project: str, content_hash: str) -> bool: + """Re-run Buckaroo's pipeline for an entry's grid (``force_reload``), for an unfaithful heal (ADR-007 D6). - Used after an unfaithful self-heal (ADR D10): the entry's snapshot bytes - changed under a stable path, so an open session's row cache would show - the old rows beside freshly computed stats. Returns whether a session - was actually evicted. The Buckaroo-side session object is left to its - own idle reaping — only the hash→session mapping is dropped here, which - is what makes the next ``load_session`` create a new one. + The snapshot's path now holds different rows and an open grid holds stats computed from the old ones. The + caller has already wiped the entry's stat cache; Buckaroo then recomputes for that session. Returns whether + Buckaroo accepted the load. Never raises: a heal must not fail because a grid could not be refreshed. """ - with self._session_lock: - removed = self._sessions.pop(content_hash, None) is not None - if removed: - self._persist_sessions() - log.info("evicted buckaroo session for %s (unfaithful heal)", content_hash) - return removed + if not self.is_running or self.bound_port is None: + return False + try: + body = self._load_body(project, content_hash, None) + except Exception as exc: + log.warning("could not build a forced reload for %s: %s", content_hash, exc) + return False + body["force_reload"] = True + try: + timeout = self._load_timeout(project, content_hash) + resp = self._client.post(f"{self.base_url}/load_expr", json=body, timeout=timeout) + resp.raise_for_status() + except httpx.HTTPError as exc: + log.warning("buckaroo forced reload failed for %s: %s", content_hash, exc) + return False + log.info("forced a reload of the grid for %s (unfaithful heal)", content_hash) + return True def _clear_stat_cache(self, project: str, content_hash: str) -> None: """Delete cached stat parquet files for one entry. @@ -516,35 +479,88 @@ def ensure_session( """ return self.load_session(content_hash, project, column_config_overrides)["session_id"] + def _load_timeout(self, project: str, content_hash: str) -> float: + row_count = 0 + mpath = entry_manifest_path(project, content_hash) + if mpath.exists(): + try: + row_count = read_manifest(mpath.parent).row_count or 0 + except Exception: + pass + return 10.0 + row_count / 1_000_000 + + def _load_body(self, project: str, content_hash: str, column_config_overrides: dict | None) -> dict: + """The ``/load_expr`` body for an entry whose files all exist. + + A worthy entry is handed a view build of its snapshot, so Buckaroo never executes an aggregate, join or sort on + tallyman's behalf and never writes a snapshot (ADR-007 D6). A cheap entry is handed its own expanded build, a + stored definition over files that exist, which tallyman already executed in full when it was created. + """ + from tallyman_xorq.portable import ensure_expanded_build # noqa: PLC0415 + from tallyman_xorq.result_cache import cache_worthy # noqa: PLC0415 + + if cache_worthy(project, content_hash): + build_dir = ensure_view_build(project, content_hash) + else: + # Expand ${TALLYMAN_PROJECT_ROOT} to absolute paths into a stable per-entry dir (not a random tmp dir) so + # the expanded path is identical across server restarts: Buckaroo's stat keys include the build + # directory's path, so a random tmp path makes every stat-cache lookup a miss even when the cache is fully + # populated on disk. Marker-gated for crash safety. + build_dir = ensure_expanded_build( + entry_build_dir(project, content_hash), + project_dir(project), + entry_expanded_build_dir(project, content_hash), + ) + stat_cache = entry_stat_cache_dir(project, content_hash) + stat_cache.mkdir(parents=True, exist_ok=True) + payload: dict = { + "session": self.session_id_for(project, content_hash), + "build_dir": str(build_dir), + "no_browser": True, + # Buckaroo scans /stats/*.py and /post_processing/*.py for project-authored + # klasses. tallyman stores both under artifacts/, so pass artifacts_dir, not project_dir. Older buckaroo + # builds ignore this field, so it's safe to always send. + "project_root": str(artifacts_dir(project)), + # Buckaroo 0.14.9+: persist computed summary stats to disk so they survive a Buckaroo restart without full + # recomputation on next /load_expr. + "cache_storage_path": str(stat_cache), + # ADR-008 D8: the column with no ties that pages sort by (buckaroo-data/buckaroo#974). A page is + # ORDER BY __row_order, or the user's keys and then __row_order, so the same request returns the same rows. + # Buckaroo builds that predate the hint ignore it. + "row_order_column": ROW_ORDER, + } + if column_config_overrides is not None: + payload["column_config_overrides"] = column_config_overrides + if self.companion_base_url is not None: + # buckaroo#943: the server fire-and-forget POSTs one record per firstpull.* perf span (expr load, stats + # pipeline + cache hit/miss, WS first payload) to this URL, keyed by the session id. Per-project so the + # receiver knows which telemetry.jsonl to append to. Older buckaroo ignores the field. + payload["telemetry_url"] = f"{self.companion_base_url}/{project}/api/telemetry" + return payload + def load_session( self, content_hash: str, project: str, column_config_overrides: dict | None = None, ) -> dict: - """Load (or reuse) a Buckaroo session, returning a typed status. + """Open (or reuse) a Buckaroo session, returning a typed status. ``{"status", "session_id", "detail"}`` where ``status`` is one of ``ok`` (``session_id`` set), ``unavailable`` (Buckaroo not running), ``no_build`` (entry has no xorq build), ``timeout`` (the ``/load_expr`` - POST timed out), or ``error`` (Buckaroo rejected the load). The companion - surfaces this so the detail page shows a spinner, a precise error, and a - retry instead of a bare "not available" fallback (#133) — never raising, - so a Buckaroo hiccup can't take down the page. - - Posts an expanded xorq build dir to Buckaroo's ``/load_expr`` endpoint so - the session is backed by an xorq expression (push-down sort/search - against the underlying backend). The posted build is always the entry's - expanded recipe (``xorq_build/``): a cheap entry recomputes on read - (push-down over a columnar source), and a cache-worthy entry's recipe - replays onto its baked ``.cache()`` snapshot — a read of the snapshot, not - a re-run of the Aggregate/Join/Sort. No per-entry ``result.parquet`` is - read any more; the #71 result-read build that served one was removed. - - ``project`` names the project that owns the entry. Sessions cache - across projects via content hash (globally unique), so a second call - for the same hash from a different project returns the existing - session — the bytes are the same by definition. + POST timed out), or ``error`` (a file the entry needs could not be made, or Buckaroo rejected the load). The + companion surfaces this so the detail page shows a spinner, a precise error, and a retry instead of a bare "not + available" fallback (#133) — never raising, so a Buckaroo hiccup can't take down the page. + + Tallyman finishes its own work first (ADR-007 D6, the governing rule): ``ensure_materialized`` makes every file + the entry's plan reads, and its own snapshot, exist and verifies what it writes. Only then is Buckaroo asked to + display anything, so a failure of the computation surfaces here, in tallyman's process, and never inside a grid + query. The session id is derived from the project and the hash and posted every time: Buckaroo skips the work + when it already holds that session with the same build dir (and the post carries none of the config-bearing + fields), and creates the session again if it has dropped it. + + ``project`` names the project that owns the entry. ``column_config_overrides`` is passed to ``/load_expr`` when provided (e.g. for promoted diff entries that carry Buckaroo coloring state). @@ -560,107 +576,54 @@ def load_session( "session_id": None, "detail": "Buckaroo is not running — start tallyman with --buckaroo.", } - build_dir = entry_build_dir(project, content_hash) - if not build_dir.is_dir(): + if not entry_build_dir(project, content_hash).is_dir(): return { "status": "no_build", "session_id": None, "detail": "This entry has no xorq build to load.", } - # No pre-heal here (ADR D4): entry builds are self-contained — a chained - # child's build carries its ancestors' cache nodes — so Buckaroo's replay - # of the build regenerates any evicted snapshot through ordinary cache - # mechanics on first query. The discarded-result cached_result_expr call - # that used to force ancestor snapshots onto disk before the replay is - # retired with the #75 stripping that made it necessary. - with self._session_lock: - cached = self._sessions.get(content_hash) - if cached: - # Within one Buckaroo lifetime cached sessions are valid by - # construction; start() resets the map on restart. - return {"status": "ok", "session_id": cached["session_id"], "detail": ""} - # Expand ${TALLYMAN_PROJECT_ROOT} to absolute paths into a stable - # per-entry dir (not a random tmp dir) so the expanded path is - # identical across server restarts — xorq embeds the build_dir path - # in the expression hash used by ParquetSnapshotCache, so a random - # tmp path makes every stat-cache lookup a miss even when the cache - # is fully populated on disk. Marker-gated for crash safety. - from tallyman_xorq.portable import ensure_expanded_build # noqa: PLC0415 - - expanded = ensure_expanded_build( - build_dir, - project_dir(project), - entry_expanded_build_dir(project, content_hash), - ) - stat_cache = entry_stat_cache_dir(project, content_hash) - stat_cache.mkdir(parents=True, exist_ok=True) - payload: dict = { - "build_dir": str(expanded), - "no_browser": True, - # Buckaroo scans /stats/*.py and - # /post_processing/*.py for project-authored - # klasses. tallyman stores both under artifacts/, so pass - # artifacts_dir, not project_dir. Older buckaroo builds - # ignore this field, so it's safe to always send. - "project_root": str(artifacts_dir(project)), - # Buckaroo 0.14.9+: persist computed summary stats to - # disk so they survive a Buckaroo restart without full - # recomputation on next /load_expr. - "cache_storage_path": str(stat_cache), + try: + from tallyman_xorq.materialize import ensure_materialized # noqa: PLC0415 + + ensure_materialized(project, content_hash) + payload = self._load_body(project, content_hash, column_config_overrides) + except Exception as exc: + log.warning("could not prepare %s for Buckaroo: %s", content_hash, exc) + return { + "status": "error", + "session_id": None, + "detail": f"Tallyman could not prepare this entry: {type(exc).__name__}: {exc}", } - if column_config_overrides is not None: - payload["column_config_overrides"] = column_config_overrides - if self.companion_base_url is not None: - # buckaroo#943: the server fire-and-forget POSTs one record per - # firstpull.* perf span (expr load, stats pipeline + cache - # hit/miss, WS first payload) to this URL, keyed by the session - # id it mints below. Per-project so the receiver knows which - # telemetry.jsonl to append to. Older buckaroo ignores the field. - payload["telemetry_url"] = f"{self.companion_base_url}/{project}/api/telemetry" - _row_count = 0 - _mpath = entry_manifest_path(project, content_hash) - if _mpath.exists(): - try: - _row_count = read_manifest(_mpath.parent).row_count or 0 - except Exception: - pass - _load_timeout = 10.0 + _row_count / 1_000_000 - _t_post = time.perf_counter() - try: - resp = self._client.post( - f"{self.base_url}/load_expr", - json=payload, - timeout=_load_timeout, - ) - resp.raise_for_status() - session_id = resp.json()["session"] - except httpx.TimeoutException as exc: - log.warning("buckaroo /load_expr timed out for %s: %s", content_hash, exc) - return { - "status": "timeout", - "session_id": None, - "detail": ( - f"Buckaroo timed out loading this entry ({_load_timeout:.1f}s) — it may be slow to materialise." - ), - } - except (httpx.HTTPError, KeyError, json.JSONDecodeError) as exc: - log.warning("buckaroo /load_expr failed for %s: %s", content_hash, exc) - return { - "status": "error", - "session_id": None, - "detail": f"Buckaroo could not load this entry: {type(exc).__name__}: {exc}", - } - self._sessions[content_hash] = {"session_id": session_id, "project": project} - self._persist_sessions() - # load_expr_ms: the companion-visible slice of the grid load — the POST - # to buckaroo only. The in-buckaroo timing (stats, row requests) needs - # buckaroo-side telemetry (buckaroo-data/buckaroo#943). + _load_timeout = self._load_timeout(project, content_hash) + _t_post = time.perf_counter() + try: + resp = self._client.post(f"{self.base_url}/load_expr", json=payload, timeout=_load_timeout) + resp.raise_for_status() + session_id = resp.json()["session"] + except httpx.TimeoutException as exc: + log.warning("buckaroo /load_expr timed out for %s: %s", content_hash, exc) return { - "status": "ok", - "session_id": session_id, - "detail": "", - "load_expr_ms": round((time.perf_counter() - _t_post) * 1000, 1), + "status": "timeout", + "session_id": None, + "detail": ( + f"Buckaroo timed out loading this entry ({_load_timeout:.1f}s) — it may be slow to materialise." + ), } + except (httpx.HTTPError, KeyError, json.JSONDecodeError) as exc: + log.warning("buckaroo /load_expr failed for %s: %s", content_hash, exc) + return { + "status": "error", + "session_id": None, + "detail": f"Buckaroo could not load this entry: {type(exc).__name__}: {exc}", + } + # load_expr_ms: the companion-visible slice of the grid load — the POST to buckaroo only. The in-buckaroo + # timing (stats, row requests) needs buckaroo-side telemetry (buckaroo-data/buckaroo#943). + return { + "status": "ok", + "session_id": session_id, + "detail": "", + "load_expr_ms": round((time.perf_counter() - _t_post) * 1000, 1), + } # ------------------------------------------------------------------ # introspection @@ -670,7 +633,6 @@ def status(self) -> dict: return { "running": self.is_running, "port": self.bound_port, - "session_count": len(self._sessions), } diff --git a/src/tallyman_companion/diff.py b/src/tallyman_companion/diff.py index f5231dd..f1d521f 100644 --- a/src/tallyman_companion/diff.py +++ b/src/tallyman_companion/diff.py @@ -17,6 +17,8 @@ from typing import Any +from tallyman_xorq.row_order import ROW_ORDER + def _is_comparable(dt_a: Any, dt_b: Any) -> bool: """True if two dtypes can be compared with ibis ``==`` without a XorqTypeError. @@ -39,8 +41,8 @@ def _classify_shared(a_schema: Any, b_schema: Any, keys: list[str]) -> tuple[lis numeric_shared ⊆ eq_shared. A name-shared column in neither set changed to an incomparable dtype and renders side-by-side with no equality term. """ - a_non_keys = [c for c in a_schema if c not in keys] - b_non_keys = [c for c in b_schema if c not in keys] + a_non_keys = [c for c in a_schema if c not in keys and c != ROW_ORDER] + b_non_keys = [c for c in b_schema if c not in keys and c != ROW_ORDER] shared = [c for c in a_non_keys if c in b_non_keys] numeric_shared = {c for c in shared if a_schema[c].is_numeric() and b_schema[c].is_numeric()} eq_shared = {c for c in shared if _is_comparable(a_schema[c], b_schema[c])} @@ -153,6 +155,11 @@ def build_compare_expr(a_expr: Any, b_expr: Any, keys: list[str]) -> tuple[Any, import xorq.vendor.ibis as ibis from buckaroo.compare import _align_backends + # A diff has no row-order column of its own from either side (ADR-008 D6): each side's positions mean nothing to + # the other, and the join would leave a ``__row_order_v2`` behind. A promoted diff is a worthy entry, since it + # contains a join, and gets its own when it is materialized. + a_expr = a_expr.drop(ROW_ORDER) if ROW_ORDER in a_expr.columns else a_expr + b_expr = b_expr.drop(ROW_ORDER) if ROW_ORDER in b_expr.columns else b_expr a_schema = a_expr.schema() b_schema = b_expr.schema() a_non_keys, b_non_keys, numeric_shared, eq_shared = _classify_shared(a_schema, b_schema, keys) diff --git a/src/tallyman_mcp/server.py b/src/tallyman_mcp/server.py index 0650d6e..20c05bc 100644 --- a/src/tallyman_mcp/server.py +++ b/src/tallyman_mcp/server.py @@ -271,8 +271,7 @@ def catalog_run(code: str, prompt: str = "") -> dict: THREE NAMESPACES — mixing these up is the #1 build failure: - import xorq.api as xo # backends + deferred reads: - # xo.memtable, xo.deferred_read_parquet, xo.connect + import xorq.api as xo # backends: xo.memtable, xo.connect import xorq.vendor.ibis as ibis # the expression API: ibis._, ibis.cases, # ibis.window, ibis.literal, ibis.coalesce, ibis.desc from tallyman_xorq.io import read_project_file, tracked_expr_from_alias, tallyman_read_csv # reading data @@ -283,7 +282,8 @@ def catalog_run(code: str, prompt: str = "") -> dict: # NEVER bare `import ibis` / `from ibis ...`: it builds but fails at save time # with a vendored-Expr class error. Always `import xorq.vendor.ibis as ibis`. # Math is a COLUMN METHOD: col.sin(), col.log(), col.sqrt() — not ibis.sin(col). - # Read data only via read_project_file / tracked_expr_from_alias (no xo.read_parquet / ibis.read_parquet). + # Read data only via read_project_file / tracked_expr_from_alias / tallyman_read_csv (no + # xo.read_parquet / xo.deferred_read_parquet / ibis.read_parquet: a raw read is a build error). ONLY BACKEND — xorq's built-in datafusion; there is NO duckdb. Do not use `ibis.duckdb`, a duckdb connection, `.sql()`, or `con.register()`. Build @@ -294,9 +294,12 @@ def catalog_run(code: str, prompt: str = "") -> dict: as a parent in the lineage DAG. Use this for normal recipe chaining — the standard way to build on top of another catalog entry. - `read_project_file("file.parquet")` reads raw files under `/data/`. - Use this only for raw files visible on disk (not catalog aliases). - - `tallyman_read_csv("/abs/path/to/file.csv", schema=...)` reads a CSV and - injects ``original_row_order`` so the snapshot is byte-stable across builds. + Use this only for raw files visible on disk (not catalog aliases). The file + is ingested once into a copy in file order with a last column, `__row_order` + (see ROW ORDER below). + - `tallyman_read_csv("/abs/path/to/file.csv", schema=...)` reads a CSV, in file + order, with a last column ``__row_order`` (see ROW ORDER below). Editing the CSV + and re-running the recipe creates a NEW entry; the old one keeps its rows. Use this for ALL CSV ingests instead of ``xo.deferred_read_csv``. - `pinned_expr_from_alias()` reads a catalog entry by content hash or explicit version reference (e.g. "shoe_sales-v2") and @@ -308,6 +311,27 @@ def catalog_run(code: str, prompt: str = "") -> dict: is often actually an alias — `read_project_file` on an alias raises `ProjectDataNotFound`. + ROW ORDER — every entry carries a last column, `__row_order`: + + `__row_order` is each row's position in the entry's file (0..N-1, int64, always + the LAST column, visible in the grid). Pages of every entry are ordered by it, + so paging is repeatable. Rules: + - A filter / select / computed column keeps it. A `select` that lists columns + and leaves it out is a BUILD ERROR (the error shows the fix), because such an + entry has no file of its own and pages by its parent's column: + t.select("region", "price", "__row_order") # not t.select("region", "price") + - An aggregate, join, sort, window function, union, distinct, unnest or UDF + makes the entry materialized: its file is written when it is created, and + `__row_order` is renumbered to match the order of the result. + - To change the order rows are shown in, sort them (`order_by`): the column is + renumbered in that order. Never assign to `__row_order` (build error). To keep + the parent's positions, copy them: `t.mutate(__row_order_v1=t["__row_order"])`. + - Joining three entries in ONE recipe: drop the column from the right-hand + inputs, `a.join(b.drop("__row_order"), k).join(c.drop("__row_order"), k)`. + Joining a join entry to another entry needs nothing. + - An `order_by` that is followed by more steps is kept, if its key columns are + still there (a build error names the key otherwise). + COLUMN NAMES — do not guess. The source step returns a `schema`, and `catalog_list` shows each entry's columns as a compact `name:type, ...` summary. Reference the exact names you see there (`tripduration`, not @@ -334,8 +358,8 @@ def catalog_run(code: str, prompt: str = "") -> dict: LOADING RAW CSV FILES — always inspect first: Use ``tallyman_read_csv`` (not ``xo.deferred_read_csv``) for all CSV - ingests. It adds an ``original_row_order`` column that makes the - snapshot byte-stable across builds: + ingests. It adds a last ``__row_order`` column holding each row's + position in the file (0..N-1): from tallyman_xorq.io import tallyman_read_csv import xorq.vendor.ibis as ibis @@ -591,6 +615,9 @@ def _run_and_record(project: str, code: str, prompt: str, *, tool: str = "catalo } if result.lint_warnings: reply["lint_warnings"] = result.lint_warnings + if result.reproducible is False: + reply["reproducible"] = False + reply["nonreproducible_columns"] = result.nonreproducible_columns return reply From b5856bb11f178ef88cbc4b24afca3d8897caebc6 Mon Sep 17 00:00:00 2001 From: Paddy Mullen Date: Mon, 21 Sep 2026 00:33:40 -0400 Subject: [PATCH 009/111] feat(app): show pinned and orphan snapshots on the Cache page The API now marks a snapshot pinned (with a reason) when it cannot be re-created faithfully, and lists a file whose entry is gone as an orphan. The page shows both, disables delete for a pinned file, and shows the reason when the server refuses. Co-Authored-By: Claude Sonnet 5 --- packages/app/src/api.ts | 8 +++++-- packages/app/src/pages/CachePage.tsx | 31 +++++++++++++++++++++------- packages/app/src/types.ts | 5 +++++ 3 files changed, 34 insertions(+), 10 deletions(-) diff --git a/packages/app/src/api.ts b/packages/app/src/api.ts index 06b640f..d49a908 100644 --- a/packages/app/src/api.ts +++ b/packages/app/src/api.ts @@ -145,8 +145,12 @@ export const api = { get(`/${project}/api/result_cache`), deleteResultCache: (project: string, hash: string): Promise<{ ok: boolean; hash: string }> => - fetch(`/${project}/api/result_cache/${hash}`, { method: "DELETE" }).then((r) => { - if (!r.ok) throw new Error(`delete failed: HTTP ${r.status}`); + fetch(`/${project}/api/result_cache/${hash}`, { method: "DELETE" }).then(async (r) => { + if (!r.ok) { + // A pinned snapshot answers 409 with the reason in `detail`; show it instead of a bare status. + const body = await r.json().catch(() => null); + throw new Error(body?.detail ?? `delete failed: HTTP ${r.status}`); + } return r.json() as Promise<{ ok: boolean; hash: string }>; }), }; diff --git a/packages/app/src/pages/CachePage.tsx b/packages/app/src/pages/CachePage.tsx index 7386497..386230d 100644 --- a/packages/app/src/pages/CachePage.tsx +++ b/packages/app/src/pages/CachePage.tsx @@ -21,13 +21,13 @@ export function CachePage() { const handleDelete = async (hash: string) => { if (!project) return; - if (!confirm(`Delete the cached snapshot for ${hash.slice(0, 12)}?\n\nThe entry, code and alias stay — it re-bakes on next view.`)) return; + if (!confirm(`Delete the snapshot for ${hash.slice(0, 12)}?\n\nThe entry, code and alias stay — the snapshot is made again and verified the next time the entry is opened.`)) return; setDeleting((s) => new Set(s).add(hash)); try { await api.deleteResultCache(project, hash); setEntries((prev) => prev.filter((e) => e.hash !== hash)); - } catch { - alert("delete failed"); + } catch (err) { + alert(err instanceof Error ? err.message : "delete failed"); } finally { setDeleting((s) => { const n = new Set(s); n.delete(hash); return n; }); } @@ -45,7 +45,8 @@ export function CachePage() { {totalFormatted} on disk - Deleting frees the snapshot only — code, build and stat cache stay; the entry re-bakes on next view. + Deleting frees the snapshot only — code, build and stat cache stay; the entry is made again and verified on next view. + A pinned snapshot cannot be made again faithfully, so it is kept. @@ -81,14 +82,27 @@ export function CachePage() { current )} + ) : e.orphan ? ( + + (no entry) + ) : ( (scratch) )} + {e.pinned && ( + + pinned + + )} - - {e.hash} - + {e.orphan ? ( + {e.hash} + ) : ( + + {e.hash} + + )} {e.prompt ?? —} @@ -96,7 +110,8 @@ export function CachePage() {